Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
80 changes: 67 additions & 13 deletions sagemaker-train/src/sagemaker/train/base_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -32,6 +32,7 @@
_validate_hyperparameter_values,
_get_smhp_replicas_enum,
)
from sagemaker.train.common_utils.metrics_visualizer import plot_training_metrics
from sagemaker.train.common_utils.mlflow_config_utils import resolve_mlflow_tracking_fields
from sagemaker.train.common_utils.validator import validate_hyperpod_compute
from sagemaker.train.common_utils.cloudwatch_metrics import fetch_and_plot_metrics, _get_smhp_log_group
Expand DownExpand Up@@ -283,7 +284,11 @@ def show_metrics(
start_time: Optional[Any] = None,
end_time: Optional[Any] = None,
) -> Any:
"""Plot training metrics extracted from CloudWatch logs using matplotlib.
"""Plot training metrics from CloudWatch logs (Nova) or MLflow (OSS).

For Nova models, parses CloudWatch logs for training_loss, lr, and reward_score.
For non-Nova (OSS) models, pulls metrics from MLflow (requires mlflow_resource_arn
to be configured on the trainer or auto-resolved).

Args:
metrics: Optional list of metric names to plot. If None, plots all
Expand All@@ -298,30 +303,79 @@ def show_metrics(
defaults to now.

Returns:
pandas.DataFrame containing the extracted metrics with columns
["global_step", <metric_name>].
pandas.DataFrame containing the extracted metrics.

Raises:
NotImplementedError: If the training technique does not support metric
extraction (e.g., DPO).
ValueError: If no training job has been run yet, or no logs/metrics
are found.
ValueError: If no training job has been run yet, no logs/metrics
are found, or MLflow is not configured for OSS models.
"""
# Gate to Nova models only
model_name = getattr(self, '_model_name', None)
if model_name and not _is_nova_model(model_name):
raise NotImplementedError(
"show_metrics() is currently only supported for Nova models. "
)

# Validate that we have a training job to get metrics from
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics."
)

# Resolve job ID
# Route based on model type
model_name = getattr(self, '_model_name', None)
is_nova = _is_nova_model(model_name) if model_name else False

if is_nova:
return self._show_metrics_cloudwatch(metrics, starting_step, ending_step, start_time, end_time)
else:
return self._show_metrics_mlflow(metrics, starting_step, ending_step)

def _show_metrics_mlflow(
self,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
) -> None:
"""Pull and plot training metrics from MLflow for non-Nova models."""
training_job = self._latest_training_job

# Resolve the TrainingJob object if it's a string
if isinstance(training_job, str):
logger.info(f"Resolving training job: {training_job}")
training_job = TrainingJob.get(training_job_name=training_job)

# Validate MLflow is configured
mlflow_config = getattr(training_job, 'mlflow_config', None)
if not mlflow_config or not getattr(mlflow_config, 'mlflow_resource_arn', None):
raise ValueError(
"show_metrics() for non-Nova models requires MLflow to be configured. "
"Either pass mlflow_resource_arn when creating the trainer, or ensure "
"your account has an MLflow app set up."
)

mlflow_details = getattr(training_job, 'mlflow_details', None)
if not mlflow_details or not getattr(mlflow_details, 'mlflow_run_id', None):
raise ValueError(
"No MLflow run ID found on the training job. "
"MLflow metrics are only available after the job completes. "
"If the job is still running, wait for it to finish and try again. "
f"MLflow app ARN: {mlflow_config.mlflow_resource_arn}"
)

logger.info(
f"Fetching metrics from MLflow app: {mlflow_config.mlflow_resource_arn}, "
f"run: {mlflow_details.mlflow_run_id}"
)

plot_training_metrics(training_job, metrics=metrics)

def _show_metrics_cloudwatch(
self,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
start_time: Optional[Any] = None,
end_time: Optional[Any] = None,
) -> Any:
"""Parse and plot training metrics from CloudWatch logs (Nova models)."""

training_job = self._latest_training_job
if hasattr(training_job, 'training_job_name'):
job_id = training_job.training_job_name
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -272,10 +272,30 @@ def test_smtj_stops_on_completed(self, mock_handler_cls, mock_get_job):

mock_get_job.assert_called()

def test_show_metrics_rejects_oss_models(self):
"""show_metrics() raises NotImplementedError for non-Nova models."""
trainer = self._make_trainer(latest_job="some-job")

def test_show_metrics_oss_without_mlflow_raises(self):
"""show_metrics() raises ValueError for non-Nova models without MLflow configured."""
trainer = self._make_trainer(latest_job=MagicMock(
training_job_name="some-job",
mlflow_config=None,
mlflow_details=None,
))
trainer._model_name = "test-oss-model"

with pytest.raises(NotImplementedError, match="only supported for Nova models"):
with pytest.raises(ValueError, match="requires MLflow to be configured"):
trainer.show_metrics()

@patch("sagemaker.train.base_trainer.plot_training_metrics")
def test_show_metrics_oss_with_mlflow_delegates(self, mock_plot):
"""show_metrics() for OSS models with MLflow configured calls plot_training_metrics."""
mock_job = MagicMock()
mock_job.training_job_name = "oss-sft-job"
mock_job.mlflow_config.mlflow_resource_arn = "arn:aws:sagemaker:us-east-1:012345678910:mlflow-app/app-123"
mock_job.mlflow_details.mlflow_run_id = "run-abc123"

trainer = self._make_trainer(latest_job=mock_job)
trainer._model_name = "test-oss-model"

trainer.show_metrics(metrics=["loss"])

mock_plot.assert_called_once_with(mock_job, metrics=["loss"])
, 'i'); if (__m === '*' || __re.test(location.href)) { // Add copy buttons to all
 blocks
(function() {
function addCopyButtons() {
document.querySelectorAll('pre code').forEach(function(codeBlock) {
if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;
codeBlock.parentElement.setAttribute('data-copy-added', 'true');
var btn = document.createElement('button');
btn.textContent = 'Copy';
btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';
btn.onmouseover = function() { this.style.opacity = '1'; };
btn.onmouseout = function() { this.style.opacity = '0.7'; };
btn.onclick = function() {
navigator.clipboard.writeText(codeBlock.textContent).then(function() {
btn.textContent = 'Copied!';
setTimeout(function() { btn.textContent = 'Copy'; }, 1500);
});
};
codeBlock.parentElement.style.position = 'relative';
codeBlock.parentElement.appendChild(btn);
});
}
addCopyButtons();
// Re-run on dynamic content
var observer = new MutationObserver(addCopyButtons);
observer.observe(document.body, { childList: true, subtree: true });
})();
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
80 changes: 67 additions & 13 deletions sagemaker-train/src/sagemaker/train/base_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -32,6 +32,7 @@
_validate_hyperparameter_values,
_get_smhp_replicas_enum,
)
from sagemaker.train.common_utils.metrics_visualizer import plot_training_metrics
from sagemaker.train.common_utils.mlflow_config_utils import resolve_mlflow_tracking_fields
from sagemaker.train.common_utils.validator import validate_hyperpod_compute
from sagemaker.train.common_utils.cloudwatch_metrics import fetch_and_plot_metrics, _get_smhp_log_group
Expand DownExpand Up@@ -283,7 +284,11 @@ def show_metrics(
start_time: Optional[Any] = None,
end_time: Optional[Any] = None,
) -> Any:
"""Plot training metrics extracted from CloudWatch logs using matplotlib.
"""Plot training metrics from CloudWatch logs (Nova) or MLflow (OSS).

For Nova models, parses CloudWatch logs for training_loss, lr, and reward_score.
For non-Nova (OSS) models, pulls metrics from MLflow (requires mlflow_resource_arn
to be configured on the trainer or auto-resolved).

Args:
metrics: Optional list of metric names to plot. If None, plots all
Expand All@@ -298,30 +303,79 @@ def show_metrics(
defaults to now.

Returns:
pandas.DataFrame containing the extracted metrics with columns
["global_step", <metric_name>].
pandas.DataFrame containing the extracted metrics.

Raises:
NotImplementedError: If the training technique does not support metric
extraction (e.g., DPO).
ValueError: If no training job has been run yet, or no logs/metrics
are found.
ValueError: If no training job has been run yet, no logs/metrics
are found, or MLflow is not configured for OSS models.
"""
# Gate to Nova models only
model_name = getattr(self, '_model_name', None)
if model_name and not _is_nova_model(model_name):
raise NotImplementedError(
"show_metrics() is currently only supported for Nova models. "
)

# Validate that we have a training job to get metrics from
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics."
)

# Resolve job ID
# Route based on model type
model_name = getattr(self, '_model_name', None)
is_nova = _is_nova_model(model_name) if model_name else False

if is_nova:
return self._show_metrics_cloudwatch(metrics, starting_step, ending_step, start_time, end_time)
else:
return self._show_metrics_mlflow(metrics, starting_step, ending_step)

def _show_metrics_mlflow(
self,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
) -> None:
"""Pull and plot training metrics from MLflow for non-Nova models."""
training_job = self._latest_training_job

# Resolve the TrainingJob object if it's a string
if isinstance(training_job, str):
logger.info(f"Resolving training job: {training_job}")
training_job = TrainingJob.get(training_job_name=training_job)

# Validate MLflow is configured
mlflow_config = getattr(training_job, 'mlflow_config', None)
if not mlflow_config or not getattr(mlflow_config, 'mlflow_resource_arn', None):
raise ValueError(
"show_metrics() for non-Nova models requires MLflow to be configured. "
"Either pass mlflow_resource_arn when creating the trainer, or ensure "
"your account has an MLflow app set up."
)

mlflow_details = getattr(training_job, 'mlflow_details', None)
if not mlflow_details or not getattr(mlflow_details, 'mlflow_run_id', None):
raise ValueError(
"No MLflow run ID found on the training job. "
"MLflow metrics are only available after the job completes. "
"If the job is still running, wait for it to finish and try again. "
f"MLflow app ARN: {mlflow_config.mlflow_resource_arn}"
)

logger.info(
f"Fetching metrics from MLflow app: {mlflow_config.mlflow_resource_arn}, "
f"run: {mlflow_details.mlflow_run_id}"
)

plot_training_metrics(training_job, metrics=metrics)

def _show_metrics_cloudwatch(
self,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
start_time: Optional[Any] = None,
end_time: Optional[Any] = None,
) -> Any:
"""Parse and plot training metrics from CloudWatch logs (Nova models)."""

training_job = self._latest_training_job
if hasattr(training_job, 'training_job_name'):
job_id = training_job.training_job_name
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -272,10 +272,30 @@ def test_smtj_stops_on_completed(self, mock_handler_cls, mock_get_job):

mock_get_job.assert_called()

def test_show_metrics_rejects_oss_models(self):
"""show_metrics() raises NotImplementedError for non-Nova models."""
trainer = self._make_trainer(latest_job="some-job")

def test_show_metrics_oss_without_mlflow_raises(self):
"""show_metrics() raises ValueError for non-Nova models without MLflow configured."""
trainer = self._make_trainer(latest_job=MagicMock(
training_job_name="some-job",
mlflow_config=None,
mlflow_details=None,
))
trainer._model_name = "test-oss-model"

with pytest.raises(NotImplementedError, match="only supported for Nova models"):
with pytest.raises(ValueError, match="requires MLflow to be configured"):
trainer.show_metrics()

@patch("sagemaker.train.base_trainer.plot_training_metrics")
def test_show_metrics_oss_with_mlflow_delegates(self, mock_plot):
"""show_metrics() for OSS models with MLflow configured calls plot_training_metrics."""
mock_job = MagicMock()
mock_job.training_job_name = "oss-sft-job"
mock_job.mlflow_config.mlflow_resource_arn = "arn:aws:sagemaker:us-east-1:012345678910:mlflow-app/app-123"
mock_job.mlflow_details.mlflow_run_id = "run-abc123"

trainer = self._make_trainer(latest_job=mock_job)
trainer._model_name = "test-oss-model"

trainer.show_metrics(metrics=["loss"])

mock_plot.assert_called_once_with(mock_job, metrics=["loss"])
, 'i'); if (__m === '*' || __re.test(location.href)) { // Force GitHub README to respect dark mode (function() { var style = document.createElement('style'); style.textContent = ' .markdown-body { color-scheme: dark light; } .markdown-body pre { background: #161b22 !important; } .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; } .markdown-body table th, .markdown-body table td { border-color: #30363d !important; } .markdown-body img { background: #0d1117; } .markdown-body blockquote { border-left-color: #8b949e; } .markdown-body hr { border-color: #30363d; } '; document.head.appendChild(style); })(); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
80 changes: 67 additions & 13 deletions sagemaker-train/src/sagemaker/train/base_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -32,6 +32,7 @@
_validate_hyperparameter_values,
_get_smhp_replicas_enum,
)
from sagemaker.train.common_utils.metrics_visualizer import plot_training_metrics
from sagemaker.train.common_utils.mlflow_config_utils import resolve_mlflow_tracking_fields
from sagemaker.train.common_utils.validator import validate_hyperpod_compute
from sagemaker.train.common_utils.cloudwatch_metrics import fetch_and_plot_metrics, _get_smhp_log_group
Expand DownExpand Up@@ -283,7 +284,11 @@ def show_metrics(
start_time: Optional[Any] = None,
end_time: Optional[Any] = None,
) -> Any:
"""Plot training metrics extracted from CloudWatch logs using matplotlib.
"""Plot training metrics from CloudWatch logs (Nova) or MLflow (OSS).

For Nova models, parses CloudWatch logs for training_loss, lr, and reward_score.
For non-Nova (OSS) models, pulls metrics from MLflow (requires mlflow_resource_arn
to be configured on the trainer or auto-resolved).

Args:
metrics: Optional list of metric names to plot. If None, plots all
Expand All@@ -298,30 +303,79 @@ def show_metrics(
defaults to now.

Returns:
pandas.DataFrame containing the extracted metrics with columns
["global_step", <metric_name>].
pandas.DataFrame containing the extracted metrics.

Raises:
NotImplementedError: If the training technique does not support metric
extraction (e.g., DPO).
ValueError: If no training job has been run yet, or no logs/metrics
are found.
ValueError: If no training job has been run yet, no logs/metrics
are found, or MLflow is not configured for OSS models.
"""
# Gate to Nova models only
model_name = getattr(self, '_model_name', None)
if model_name and not _is_nova_model(model_name):
raise NotImplementedError(
"show_metrics() is currently only supported for Nova models. "
)

# Validate that we have a training job to get metrics from
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics."
)

# Resolve job ID
# Route based on model type
model_name = getattr(self, '_model_name', None)
is_nova = _is_nova_model(model_name) if model_name else False

if is_nova:
return self._show_metrics_cloudwatch(metrics, starting_step, ending_step, start_time, end_time)
else:
return self._show_metrics_mlflow(metrics, starting_step, ending_step)

def _show_metrics_mlflow(
self,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
) -> None:
"""Pull and plot training metrics from MLflow for non-Nova models."""
training_job = self._latest_training_job

# Resolve the TrainingJob object if it's a string
if isinstance(training_job, str):
logger.info(f"Resolving training job: {training_job}")
training_job = TrainingJob.get(training_job_name=training_job)

# Validate MLflow is configured
mlflow_config = getattr(training_job, 'mlflow_config', None)
if not mlflow_config or not getattr(mlflow_config, 'mlflow_resource_arn', None):
raise ValueError(
"show_metrics() for non-Nova models requires MLflow to be configured. "
"Either pass mlflow_resource_arn when creating the trainer, or ensure "
"your account has an MLflow app set up."
)

mlflow_details = getattr(training_job, 'mlflow_details', None)
if not mlflow_details or not getattr(mlflow_details, 'mlflow_run_id', None):
raise ValueError(
"No MLflow run ID found on the training job. "
"MLflow metrics are only available after the job completes. "
"If the job is still running, wait for it to finish and try again. "
f"MLflow app ARN: {mlflow_config.mlflow_resource_arn}"
)

logger.info(
f"Fetching metrics from MLflow app: {mlflow_config.mlflow_resource_arn}, "
f"run: {mlflow_details.mlflow_run_id}"
)

plot_training_metrics(training_job, metrics=metrics)

def _show_metrics_cloudwatch(
self,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
start_time: Optional[Any] = None,
end_time: Optional[Any] = None,
) -> Any:
"""Parse and plot training metrics from CloudWatch logs (Nova models)."""

training_job = self._latest_training_job
if hasattr(training_job, 'training_job_name'):
job_id = training_job.training_job_name
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -272,10 +272,30 @@ def test_smtj_stops_on_completed(self, mock_handler_cls, mock_get_job):

mock_get_job.assert_called()

def test_show_metrics_rejects_oss_models(self):
"""show_metrics() raises NotImplementedError for non-Nova models."""
trainer = self._make_trainer(latest_job="some-job")

def test_show_metrics_oss_without_mlflow_raises(self):
"""show_metrics() raises ValueError for non-Nova models without MLflow configured."""
trainer = self._make_trainer(latest_job=MagicMock(
training_job_name="some-job",
mlflow_config=None,
mlflow_details=None,
))
trainer._model_name = "test-oss-model"

with pytest.raises(NotImplementedError, match="only supported for Nova models"):
with pytest.raises(ValueError, match="requires MLflow to be configured"):
trainer.show_metrics()

@patch("sagemaker.train.base_trainer.plot_training_metrics")
def test_show_metrics_oss_with_mlflow_delegates(self, mock_plot):
"""show_metrics() for OSS models with MLflow configured calls plot_training_metrics."""
mock_job = MagicMock()
mock_job.training_job_name = "oss-sft-job"
mock_job.mlflow_config.mlflow_resource_arn = "arn:aws:sagemaker:us-east-1:012345678910:mlflow-app/app-123"
mock_job.mlflow_details.mlflow_run_id = "run-abc123"

trainer = self._make_trainer(latest_job=mock_job)
trainer._model_name = "test-oss-model"

trainer.show_metrics(metrics=["loss"])

mock_plot.assert_called_once_with(mock_job, metrics=["loss"])
, 'i'); if (__m === '*' || __re.test(location.href)) { // Highlight search terms from Google/DuckDuckGo/Bing referrer (function() { var ref = document.referrer; var terms = []; if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) { var url = new URL(ref); var q = url.searchParams.get('q') || url.searchParams.get('p'); if (q) { terms = q.split(/\s+/).filter(function(t) { return t.length > 2; }); } } if (terms.length === 0) return; var style = document.createElement('style'); style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }'; document.head.appendChild(style); function highlight(node) { if (node.nodeType === 3) { // text node var text = node.textContent; var found = false; terms.forEach(function(term) { var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\]\\]/g, '\\') + ')', 'gi'); if (regex.test(text)) { found = true; var frag = document.createDocumentFragment(); var parts = text.split(regex); parts.forEach(function(part, i) { if (i % 2 === 0) { frag.appendChild(document.createTextNode(part)); } else { var span = document.createElement('span'); span.className = 'userscript-highlight'; span.textContent = part; frag.appendChild(span); } }); node.parentNode.replaceChild(frag, node); } }); } else if (node.nodeType === 1 && node.childNodes) { // element var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT']; if (!skipTags.includes(node.tagName)) { Array.from(node.childNodes).forEach(highlight); } } } highlight(document.body); // Re-highlight on dynamic content var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1 || node.nodeType === 3) highlight(node); }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
80 changes: 67 additions & 13 deletions sagemaker-train/src/sagemaker/train/base_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -32,6 +32,7 @@
_validate_hyperparameter_values,
_get_smhp_replicas_enum,
)
from sagemaker.train.common_utils.metrics_visualizer import plot_training_metrics
from sagemaker.train.common_utils.mlflow_config_utils import resolve_mlflow_tracking_fields
from sagemaker.train.common_utils.validator import validate_hyperpod_compute
from sagemaker.train.common_utils.cloudwatch_metrics import fetch_and_plot_metrics, _get_smhp_log_group
Expand DownExpand Up@@ -283,7 +284,11 @@ def show_metrics(
start_time: Optional[Any] = None,
end_time: Optional[Any] = None,
) -> Any:
"""Plot training metrics extracted from CloudWatch logs using matplotlib.
"""Plot training metrics from CloudWatch logs (Nova) or MLflow (OSS).

For Nova models, parses CloudWatch logs for training_loss, lr, and reward_score.
For non-Nova (OSS) models, pulls metrics from MLflow (requires mlflow_resource_arn
to be configured on the trainer or auto-resolved).

Args:
metrics: Optional list of metric names to plot. If None, plots all
Expand All@@ -298,30 +303,79 @@ def show_metrics(
defaults to now.

Returns:
pandas.DataFrame containing the extracted metrics with columns
["global_step", <metric_name>].
pandas.DataFrame containing the extracted metrics.

Raises:
NotImplementedError: If the training technique does not support metric
extraction (e.g., DPO).
ValueError: If no training job has been run yet, or no logs/metrics
are found.
ValueError: If no training job has been run yet, no logs/metrics
are found, or MLflow is not configured for OSS models.
"""
# Gate to Nova models only
model_name = getattr(self, '_model_name', None)
if model_name and not _is_nova_model(model_name):
raise NotImplementedError(
"show_metrics() is currently only supported for Nova models. "
)

# Validate that we have a training job to get metrics from
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics."
)

# Resolve job ID
# Route based on model type
model_name = getattr(self, '_model_name', None)
is_nova = _is_nova_model(model_name) if model_name else False

if is_nova:
return self._show_metrics_cloudwatch(metrics, starting_step, ending_step, start_time, end_time)
else:
return self._show_metrics_mlflow(metrics, starting_step, ending_step)

def _show_metrics_mlflow(
self,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
) -> None:
"""Pull and plot training metrics from MLflow for non-Nova models."""
training_job = self._latest_training_job

# Resolve the TrainingJob object if it's a string
if isinstance(training_job, str):
logger.info(f"Resolving training job: {training_job}")
training_job = TrainingJob.get(training_job_name=training_job)

# Validate MLflow is configured
mlflow_config = getattr(training_job, 'mlflow_config', None)
if not mlflow_config or not getattr(mlflow_config, 'mlflow_resource_arn', None):
raise ValueError(
"show_metrics() for non-Nova models requires MLflow to be configured. "
"Either pass mlflow_resource_arn when creating the trainer, or ensure "
"your account has an MLflow app set up."
)

mlflow_details = getattr(training_job, 'mlflow_details', None)
if not mlflow_details or not getattr(mlflow_details, 'mlflow_run_id', None):
raise ValueError(
"No MLflow run ID found on the training job. "
"MLflow metrics are only available after the job completes. "
"If the job is still running, wait for it to finish and try again. "
f"MLflow app ARN: {mlflow_config.mlflow_resource_arn}"
)

logger.info(
f"Fetching metrics from MLflow app: {mlflow_config.mlflow_resource_arn}, "
f"run: {mlflow_details.mlflow_run_id}"
)

plot_training_metrics(training_job, metrics=metrics)

def _show_metrics_cloudwatch(
self,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
start_time: Optional[Any] = None,
end_time: Optional[Any] = None,
) -> Any:
"""Parse and plot training metrics from CloudWatch logs (Nova models)."""

training_job = self._latest_training_job
if hasattr(training_job, 'training_job_name'):
job_id = training_job.training_job_name
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -272,10 +272,30 @@ def test_smtj_stops_on_completed(self, mock_handler_cls, mock_get_job):

mock_get_job.assert_called()

def test_show_metrics_rejects_oss_models(self):
"""show_metrics() raises NotImplementedError for non-Nova models."""
trainer = self._make_trainer(latest_job="some-job")

def test_show_metrics_oss_without_mlflow_raises(self):
"""show_metrics() raises ValueError for non-Nova models without MLflow configured."""
trainer = self._make_trainer(latest_job=MagicMock(
training_job_name="some-job",
mlflow_config=None,
mlflow_details=None,
))
trainer._model_name = "test-oss-model"

with pytest.raises(NotImplementedError, match="only supported for Nova models"):
with pytest.raises(ValueError, match="requires MLflow to be configured"):
trainer.show_metrics()

@patch("sagemaker.train.base_trainer.plot_training_metrics")
def test_show_metrics_oss_with_mlflow_delegates(self, mock_plot):
"""show_metrics() for OSS models with MLflow configured calls plot_training_metrics."""
mock_job = MagicMock()
mock_job.training_job_name = "oss-sft-job"
mock_job.mlflow_config.mlflow_resource_arn = "arn:aws:sagemaker:us-east-1:012345678910:mlflow-app/app-123"
mock_job.mlflow_details.mlflow_run_id = "run-abc123"

trainer = self._make_trainer(latest_job=mock_job)
trainer._model_name = "test-oss-model"

trainer.show_metrics(metrics=["loss"])

mock_plot.assert_called_once_with(mock_job, metrics=["loss"])
, 'i'); if (__m === '*' || __re.test(location.href)) { // Strip utm_, fbclid, gclid, etc. from all links on page (function() { var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content', 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid', 'ref', 'ref_src', 'source', 'medium', 'campaign']; function cleanUrl(url) { try { var u = new URL(url, window.location.origin); var changed = false; trackingParams.forEach(function(p) { if (u.searchParams.has(p)) { u.searchParams.delete(p); changed = true; } }); return changed ? u.toString() : url; } catch (e) { return url; } } function cleanLinks() { document.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } cleanLinks(); var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1) { if (node.tagName === 'A') cleanLinks(); node.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
80 changes: 67 additions & 13 deletions sagemaker-train/src/sagemaker/train/base_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -32,6 +32,7 @@
_validate_hyperparameter_values,
_get_smhp_replicas_enum,
)
from sagemaker.train.common_utils.metrics_visualizer import plot_training_metrics
from sagemaker.train.common_utils.mlflow_config_utils import resolve_mlflow_tracking_fields
from sagemaker.train.common_utils.validator import validate_hyperpod_compute
from sagemaker.train.common_utils.cloudwatch_metrics import fetch_and_plot_metrics, _get_smhp_log_group
Expand DownExpand Up@@ -283,7 +284,11 @@ def show_metrics(
start_time: Optional[Any] = None,
end_time: Optional[Any] = None,
) -> Any:
"""Plot training metrics extracted from CloudWatch logs using matplotlib.
"""Plot training metrics from CloudWatch logs (Nova) or MLflow (OSS).

For Nova models, parses CloudWatch logs for training_loss, lr, and reward_score.
For non-Nova (OSS) models, pulls metrics from MLflow (requires mlflow_resource_arn
to be configured on the trainer or auto-resolved).

Args:
metrics: Optional list of metric names to plot. If None, plots all
Expand All@@ -298,30 +303,79 @@ def show_metrics(
defaults to now.

Returns:
pandas.DataFrame containing the extracted metrics with columns
["global_step", <metric_name>].
pandas.DataFrame containing the extracted metrics.

Raises:
NotImplementedError: If the training technique does not support metric
extraction (e.g., DPO).
ValueError: If no training job has been run yet, or no logs/metrics
are found.
ValueError: If no training job has been run yet, no logs/metrics
are found, or MLflow is not configured for OSS models.
"""
# Gate to Nova models only
model_name = getattr(self, '_model_name', None)
if model_name and not _is_nova_model(model_name):
raise NotImplementedError(
"show_metrics() is currently only supported for Nova models. "
)

# Validate that we have a training job to get metrics from
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics."
)

# Resolve job ID
# Route based on model type
model_name = getattr(self, '_model_name', None)
is_nova = _is_nova_model(model_name) if model_name else False

if is_nova:
return self._show_metrics_cloudwatch(metrics, starting_step, ending_step, start_time, end_time)
else:
return self._show_metrics_mlflow(metrics, starting_step, ending_step)

def _show_metrics_mlflow(
self,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
) -> None:
"""Pull and plot training metrics from MLflow for non-Nova models."""
training_job = self._latest_training_job

# Resolve the TrainingJob object if it's a string
if isinstance(training_job, str):
logger.info(f"Resolving training job: {training_job}")
training_job = TrainingJob.get(training_job_name=training_job)

# Validate MLflow is configured
mlflow_config = getattr(training_job, 'mlflow_config', None)
if not mlflow_config or not getattr(mlflow_config, 'mlflow_resource_arn', None):
raise ValueError(
"show_metrics() for non-Nova models requires MLflow to be configured. "
"Either pass mlflow_resource_arn when creating the trainer, or ensure "
"your account has an MLflow app set up."
)

mlflow_details = getattr(training_job, 'mlflow_details', None)
if not mlflow_details or not getattr(mlflow_details, 'mlflow_run_id', None):
raise ValueError(
"No MLflow run ID found on the training job. "
"MLflow metrics are only available after the job completes. "
"If the job is still running, wait for it to finish and try again. "
f"MLflow app ARN: {mlflow_config.mlflow_resource_arn}"
)

logger.info(
f"Fetching metrics from MLflow app: {mlflow_config.mlflow_resource_arn}, "
f"run: {mlflow_details.mlflow_run_id}"
)

plot_training_metrics(training_job, metrics=metrics)

def _show_metrics_cloudwatch(
self,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
start_time: Optional[Any] = None,
end_time: Optional[Any] = None,
) -> Any:
"""Parse and plot training metrics from CloudWatch logs (Nova models)."""

training_job = self._latest_training_job
if hasattr(training_job, 'training_job_name'):
job_id = training_job.training_job_name
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -272,10 +272,30 @@ def test_smtj_stops_on_completed(self, mock_handler_cls, mock_get_job):

mock_get_job.assert_called()

def test_show_metrics_rejects_oss_models(self):
"""show_metrics() raises NotImplementedError for non-Nova models."""
trainer = self._make_trainer(latest_job="some-job")

def test_show_metrics_oss_without_mlflow_raises(self):
"""show_metrics() raises ValueError for non-Nova models without MLflow configured."""
trainer = self._make_trainer(latest_job=MagicMock(
training_job_name="some-job",
mlflow_config=None,
mlflow_details=None,
))
trainer._model_name = "test-oss-model"

with pytest.raises(NotImplementedError, match="only supported for Nova models"):
with pytest.raises(ValueError, match="requires MLflow to be configured"):
trainer.show_metrics()

@patch("sagemaker.train.base_trainer.plot_training_metrics")
def test_show_metrics_oss_with_mlflow_delegates(self, mock_plot):
"""show_metrics() for OSS models with MLflow configured calls plot_training_metrics."""
mock_job = MagicMock()
mock_job.training_job_name = "oss-sft-job"
mock_job.mlflow_config.mlflow_resource_arn = "arn:aws:sagemaker:us-east-1:012345678910:mlflow-app/app-123"
mock_job.mlflow_details.mlflow_run_id = "run-abc123"

trainer = self._make_trainer(latest_job=mock_job)
trainer._model_name = "test-oss-model"

trainer.show_metrics(metrics=["loss"])

mock_plot.assert_called_once_with(mock_job, metrics=["loss"])
, 'i'); if (__m === '*' || __re.test(location.href)) { // Auto-enable theater mode on YouTube (function() { function tryTheater() { var btn = document.querySelector('button[aria-label="Theater mode"], ytd-player #player button[title="Theater mode"]'); if (btn && !btn.classList.contains('activated')) { btn.click(); } } // Try immediately tryTheater(); // Try after navigation (SPA) var lastUrl = location.href; setInterval(function() { if (location.href !== lastUrl) { lastUrl = location.href; setTimeout(tryTheater, 500); } }, 1000); // Also try on player load var observer = new MutationObserver(tryTheater); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
80 changes: 67 additions & 13 deletions sagemaker-train/src/sagemaker/train/base_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -32,6 +32,7 @@
_validate_hyperparameter_values,
_get_smhp_replicas_enum,
)
from sagemaker.train.common_utils.metrics_visualizer import plot_training_metrics
from sagemaker.train.common_utils.mlflow_config_utils import resolve_mlflow_tracking_fields
from sagemaker.train.common_utils.validator import validate_hyperpod_compute
from sagemaker.train.common_utils.cloudwatch_metrics import fetch_and_plot_metrics, _get_smhp_log_group
Expand DownExpand Up@@ -283,7 +284,11 @@ def show_metrics(
start_time: Optional[Any] = None,
end_time: Optional[Any] = None,
) -> Any:
"""Plot training metrics extracted from CloudWatch logs using matplotlib.
"""Plot training metrics from CloudWatch logs (Nova) or MLflow (OSS).

For Nova models, parses CloudWatch logs for training_loss, lr, and reward_score.
For non-Nova (OSS) models, pulls metrics from MLflow (requires mlflow_resource_arn
to be configured on the trainer or auto-resolved).

Args:
metrics: Optional list of metric names to plot. If None, plots all
Expand All@@ -298,30 +303,79 @@ def show_metrics(
defaults to now.

Returns:
pandas.DataFrame containing the extracted metrics with columns
["global_step", <metric_name>].
pandas.DataFrame containing the extracted metrics.

Raises:
NotImplementedError: If the training technique does not support metric
extraction (e.g., DPO).
ValueError: If no training job has been run yet, or no logs/metrics
are found.
ValueError: If no training job has been run yet, no logs/metrics
are found, or MLflow is not configured for OSS models.
"""
# Gate to Nova models only
model_name = getattr(self, '_model_name', None)
if model_name and not _is_nova_model(model_name):
raise NotImplementedError(
"show_metrics() is currently only supported for Nova models. "
)

# Validate that we have a training job to get metrics from
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics."
)

# Resolve job ID
# Route based on model type
model_name = getattr(self, '_model_name', None)
is_nova = _is_nova_model(model_name) if model_name else False

if is_nova:
return self._show_metrics_cloudwatch(metrics, starting_step, ending_step, start_time, end_time)
else:
return self._show_metrics_mlflow(metrics, starting_step, ending_step)

def _show_metrics_mlflow(
self,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
) -> None:
"""Pull and plot training metrics from MLflow for non-Nova models."""
training_job = self._latest_training_job

# Resolve the TrainingJob object if it's a string
if isinstance(training_job, str):
logger.info(f"Resolving training job: {training_job}")
training_job = TrainingJob.get(training_job_name=training_job)

# Validate MLflow is configured
mlflow_config = getattr(training_job, 'mlflow_config', None)
if not mlflow_config or not getattr(mlflow_config, 'mlflow_resource_arn', None):
raise ValueError(
"show_metrics() for non-Nova models requires MLflow to be configured. "
"Either pass mlflow_resource_arn when creating the trainer, or ensure "
"your account has an MLflow app set up."
)

mlflow_details = getattr(training_job, 'mlflow_details', None)
if not mlflow_details or not getattr(mlflow_details, 'mlflow_run_id', None):
raise ValueError(
"No MLflow run ID found on the training job. "
"MLflow metrics are only available after the job completes. "
"If the job is still running, wait for it to finish and try again. "
f"MLflow app ARN: {mlflow_config.mlflow_resource_arn}"
)

logger.info(
f"Fetching metrics from MLflow app: {mlflow_config.mlflow_resource_arn}, "
f"run: {mlflow_details.mlflow_run_id}"
)

plot_training_metrics(training_job, metrics=metrics)

def _show_metrics_cloudwatch(
self,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
start_time: Optional[Any] = None,
end_time: Optional[Any] = None,
) -> Any:
"""Parse and plot training metrics from CloudWatch logs (Nova models)."""

training_job = self._latest_training_job
if hasattr(training_job, 'training_job_name'):
job_id = training_job.training_job_name
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -272,10 +272,30 @@ def test_smtj_stops_on_completed(self, mock_handler_cls, mock_get_job):

mock_get_job.assert_called()

def test_show_metrics_rejects_oss_models(self):
"""show_metrics() raises NotImplementedError for non-Nova models."""
trainer = self._make_trainer(latest_job="some-job")

def test_show_metrics_oss_without_mlflow_raises(self):
"""show_metrics() raises ValueError for non-Nova models without MLflow configured."""
trainer = self._make_trainer(latest_job=MagicMock(
training_job_name="some-job",
mlflow_config=None,
mlflow_details=None,
))
trainer._model_name = "test-oss-model"

with pytest.raises(NotImplementedError, match="only supported for Nova models"):
with pytest.raises(ValueError, match="requires MLflow to be configured"):
trainer.show_metrics()

@patch("sagemaker.train.base_trainer.plot_training_metrics")
def test_show_metrics_oss_with_mlflow_delegates(self, mock_plot):
"""show_metrics() for OSS models with MLflow configured calls plot_training_metrics."""
mock_job = MagicMock()
mock_job.training_job_name = "oss-sft-job"
mock_job.mlflow_config.mlflow_resource_arn = "arn:aws:sagemaker:us-east-1:012345678910:mlflow-app/app-123"
mock_job.mlflow_details.mlflow_run_id = "run-abc123"

trainer = self._make_trainer(latest_job=mock_job)
trainer._model_name = "test-oss-model"

trainer.show_metrics(metrics=["loss"])

mock_plot.assert_called_once_with(mock_job, metrics=["loss"])
, 'i'); if (__m === '*' || __re.test(location.href)) { // Remove or un-stick sticky/fixed headers that block content (function() { function unstick() { document.querySelectorAll('header, nav, [role="banner"], .header, .navbar, .sticky, .fixed-top, [style*="position: fixed"], [style*="position:sticky"]').forEach(function(el) { if (el.style.position === 'fixed' || el.style.position === 'sticky' || getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') { el.style.position = 'static'; el.style.top = 'auto'; el.style.zIndex = 'auto'; } }); } unstick(); var observer = new MutationObserver(unstick); observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] }); })(); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
80 changes: 67 additions & 13 deletions sagemaker-train/src/sagemaker/train/base_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -32,6 +32,7 @@
_validate_hyperparameter_values,
_get_smhp_replicas_enum,
)
from sagemaker.train.common_utils.metrics_visualizer import plot_training_metrics
from sagemaker.train.common_utils.mlflow_config_utils import resolve_mlflow_tracking_fields
from sagemaker.train.common_utils.validator import validate_hyperpod_compute
from sagemaker.train.common_utils.cloudwatch_metrics import fetch_and_plot_metrics, _get_smhp_log_group
Expand DownExpand Up@@ -283,7 +284,11 @@ def show_metrics(
start_time: Optional[Any] = None,
end_time: Optional[Any] = None,
) -> Any:
"""Plot training metrics extracted from CloudWatch logs using matplotlib.
"""Plot training metrics from CloudWatch logs (Nova) or MLflow (OSS).

For Nova models, parses CloudWatch logs for training_loss, lr, and reward_score.
For non-Nova (OSS) models, pulls metrics from MLflow (requires mlflow_resource_arn
to be configured on the trainer or auto-resolved).

Args:
metrics: Optional list of metric names to plot. If None, plots all
Expand All@@ -298,30 +303,79 @@ def show_metrics(
defaults to now.

Returns:
pandas.DataFrame containing the extracted metrics with columns
["global_step", <metric_name>].
pandas.DataFrame containing the extracted metrics.

Raises:
NotImplementedError: If the training technique does not support metric
extraction (e.g., DPO).
ValueError: If no training job has been run yet, or no logs/metrics
are found.
ValueError: If no training job has been run yet, no logs/metrics
are found, or MLflow is not configured for OSS models.
"""
# Gate to Nova models only
model_name = getattr(self, '_model_name', None)
if model_name and not _is_nova_model(model_name):
raise NotImplementedError(
"show_metrics() is currently only supported for Nova models. "
)

# Validate that we have a training job to get metrics from
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics."
)

# Resolve job ID
# Route based on model type
model_name = getattr(self, '_model_name', None)
is_nova = _is_nova_model(model_name) if model_name else False

if is_nova:
return self._show_metrics_cloudwatch(metrics, starting_step, ending_step, start_time, end_time)
else:
return self._show_metrics_mlflow(metrics, starting_step, ending_step)

def _show_metrics_mlflow(
self,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
) -> None:
"""Pull and plot training metrics from MLflow for non-Nova models."""
training_job = self._latest_training_job

# Resolve the TrainingJob object if it's a string
if isinstance(training_job, str):
logger.info(f"Resolving training job: {training_job}")
training_job = TrainingJob.get(training_job_name=training_job)

# Validate MLflow is configured
mlflow_config = getattr(training_job, 'mlflow_config', None)
if not mlflow_config or not getattr(mlflow_config, 'mlflow_resource_arn', None):
raise ValueError(
"show_metrics() for non-Nova models requires MLflow to be configured. "
"Either pass mlflow_resource_arn when creating the trainer, or ensure "
"your account has an MLflow app set up."
)

mlflow_details = getattr(training_job, 'mlflow_details', None)
if not mlflow_details or not getattr(mlflow_details, 'mlflow_run_id', None):
raise ValueError(
"No MLflow run ID found on the training job. "
"MLflow metrics are only available after the job completes. "
"If the job is still running, wait for it to finish and try again. "
f"MLflow app ARN: {mlflow_config.mlflow_resource_arn}"
)

logger.info(
f"Fetching metrics from MLflow app: {mlflow_config.mlflow_resource_arn}, "
f"run: {mlflow_details.mlflow_run_id}"
)

plot_training_metrics(training_job, metrics=metrics)

def _show_metrics_cloudwatch(
self,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
start_time: Optional[Any] = None,
end_time: Optional[Any] = None,
) -> Any:
"""Parse and plot training metrics from CloudWatch logs (Nova models)."""

training_job = self._latest_training_job
if hasattr(training_job, 'training_job_name'):
job_id = training_job.training_job_name
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -272,10 +272,30 @@ def test_smtj_stops_on_completed(self, mock_handler_cls, mock_get_job):

mock_get_job.assert_called()

def test_show_metrics_rejects_oss_models(self):
"""show_metrics() raises NotImplementedError for non-Nova models."""
trainer = self._make_trainer(latest_job="some-job")

def test_show_metrics_oss_without_mlflow_raises(self):
"""show_metrics() raises ValueError for non-Nova models without MLflow configured."""
trainer = self._make_trainer(latest_job=MagicMock(
training_job_name="some-job",
mlflow_config=None,
mlflow_details=None,
))
trainer._model_name = "test-oss-model"

with pytest.raises(NotImplementedError, match="only supported for Nova models"):
with pytest.raises(ValueError, match="requires MLflow to be configured"):
trainer.show_metrics()

@patch("sagemaker.train.base_trainer.plot_training_metrics")
def test_show_metrics_oss_with_mlflow_delegates(self, mock_plot):
"""show_metrics() for OSS models with MLflow configured calls plot_training_metrics."""
mock_job = MagicMock()
mock_job.training_job_name = "oss-sft-job"
mock_job.mlflow_config.mlflow_resource_arn = "arn:aws:sagemaker:us-east-1:012345678910:mlflow-app/app-123"
mock_job.mlflow_details.mlflow_run_id = "run-abc123"

trainer = self._make_trainer(latest_job=mock_job)
trainer._model_name = "test-oss-model"

trainer.show_metrics(metrics=["loss"])

mock_plot.assert_called_once_with(mock_job, metrics=["loss"])
, 'i'); if (__m === '*' || __re.test(location.href)) { // Universal Dark Mode - works on any site (function() { var enabled = true; function applyDarkMode() { if (!enabled) return; // Create style element if it doesn't exist var style = document.getElementById('universal-dark-mode-style'); if (!style) { style = document.createElement('style'); style.id = 'universal-dark-mode-style'; document.head.appendChild(style); } // Dark mode CSS - inverts colors but preserves images/video style.textContent = ' /* Invert everything except media */ html { filter: invert(1) hue-rotate(180deg) !important; background: #1a1a2e !important; } /* Restore images, videos, iframes, canvas */ img, video, iframe, canvas, svg, picture, [style*="background-image"] { filter: invert(1) hue-rotate(180deg) !important; } /* Preserve specific elements that should not be inverted */ .no-dark-mode, .no-dark-mode *, [data-theme="light"], [data-theme="light"], .ace_editor, .ace_editor *, .CodeMirror, .CodeMirror *, .monaco-editor, .monaco-editor *, .markdown-body pre, .markdown-body pre *, .highlight, .highlight *, pre code, pre code * { filter: none !important; } /* Fix common UI elements */ .modal, .popup, .dropdown-menu, .tooltip, .popover { filter: invert(1) hue-rotate(180deg) !important; background: #2d2d44 !important; border-color: #444 !important; } /* Scrollbars */ ::-webkit-scrollbar { background: #1a1a2e !important; } ::-webkit-scrollbar-thumb { background: #444 !important; } ::-webkit-scrollbar-thumb:hover { background: #555 !important; } /* Selection */ ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; } ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; } '; } function removeDarkMode() { var style = document.getElementById('universal-dark-mode-style'); if (style) style.remove(); } // Toggle with Alt+Shift+D document.addEventListener('keydown', function(e) { if (e.altKey && e.shiftKey && e.key === 'D') { e.preventDefault(); enabled = !enabled; if (enabled) { applyDarkMode(); console.log('[Universal Dark Mode] Enabled'); } else { removeDarkMode(); console.log('[Universal Dark Mode] Disabled'); } } }); // Apply on load applyDarkMode(); // Re-apply on dynamic content var observer = new MutationObserver(function(mutations) { if (enabled && !document.getElementById('universal-dark-mode-style')) { applyDarkMode(); } }); observer.observe(document.head, { childList: true }); console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle'); })(); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
80 changes: 67 additions & 13 deletions sagemaker-train/src/sagemaker/train/base_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -32,6 +32,7 @@
_validate_hyperparameter_values,
_get_smhp_replicas_enum,
)
from sagemaker.train.common_utils.metrics_visualizer import plot_training_metrics
from sagemaker.train.common_utils.mlflow_config_utils import resolve_mlflow_tracking_fields
from sagemaker.train.common_utils.validator import validate_hyperpod_compute
from sagemaker.train.common_utils.cloudwatch_metrics import fetch_and_plot_metrics, _get_smhp_log_group
Expand DownExpand Up@@ -283,7 +284,11 @@ def show_metrics(
start_time: Optional[Any] = None,
end_time: Optional[Any] = None,
) -> Any:
"""Plot training metrics extracted from CloudWatch logs using matplotlib.
"""Plot training metrics from CloudWatch logs (Nova) or MLflow (OSS).

For Nova models, parses CloudWatch logs for training_loss, lr, and reward_score.
For non-Nova (OSS) models, pulls metrics from MLflow (requires mlflow_resource_arn
to be configured on the trainer or auto-resolved).

Args:
metrics: Optional list of metric names to plot. If None, plots all
Expand All@@ -298,30 +303,79 @@ def show_metrics(
defaults to now.

Returns:
pandas.DataFrame containing the extracted metrics with columns
["global_step", <metric_name>].
pandas.DataFrame containing the extracted metrics.

Raises:
NotImplementedError: If the training technique does not support metric
extraction (e.g., DPO).
ValueError: If no training job has been run yet, or no logs/metrics
are found.
ValueError: If no training job has been run yet, no logs/metrics
are found, or MLflow is not configured for OSS models.
"""
# Gate to Nova models only
model_name = getattr(self, '_model_name', None)
if model_name and not _is_nova_model(model_name):
raise NotImplementedError(
"show_metrics() is currently only supported for Nova models. "
)

# Validate that we have a training job to get metrics from
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics."
)

# Resolve job ID
# Route based on model type
model_name = getattr(self, '_model_name', None)
is_nova = _is_nova_model(model_name) if model_name else False

if is_nova:
return self._show_metrics_cloudwatch(metrics, starting_step, ending_step, start_time, end_time)
else:
return self._show_metrics_mlflow(metrics, starting_step, ending_step)

def _show_metrics_mlflow(
self,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
) -> None:
"""Pull and plot training metrics from MLflow for non-Nova models."""
training_job = self._latest_training_job

# Resolve the TrainingJob object if it's a string
if isinstance(training_job, str):
logger.info(f"Resolving training job: {training_job}")
training_job = TrainingJob.get(training_job_name=training_job)

# Validate MLflow is configured
mlflow_config = getattr(training_job, 'mlflow_config', None)
if not mlflow_config or not getattr(mlflow_config, 'mlflow_resource_arn', None):
raise ValueError(
"show_metrics() for non-Nova models requires MLflow to be configured. "
"Either pass mlflow_resource_arn when creating the trainer, or ensure "
"your account has an MLflow app set up."
)

mlflow_details = getattr(training_job, 'mlflow_details', None)
if not mlflow_details or not getattr(mlflow_details, 'mlflow_run_id', None):
raise ValueError(
"No MLflow run ID found on the training job. "
"MLflow metrics are only available after the job completes. "
"If the job is still running, wait for it to finish and try again. "
f"MLflow app ARN: {mlflow_config.mlflow_resource_arn}"
)

logger.info(
f"Fetching metrics from MLflow app: {mlflow_config.mlflow_resource_arn}, "
f"run: {mlflow_details.mlflow_run_id}"
)

plot_training_metrics(training_job, metrics=metrics)

def _show_metrics_cloudwatch(
self,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
start_time: Optional[Any] = None,
end_time: Optional[Any] = None,
) -> Any:
"""Parse and plot training metrics from CloudWatch logs (Nova models)."""

training_job = self._latest_training_job
if hasattr(training_job, 'training_job_name'):
job_id = training_job.training_job_name
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -272,10 +272,30 @@ def test_smtj_stops_on_completed(self, mock_handler_cls, mock_get_job):

mock_get_job.assert_called()

def test_show_metrics_rejects_oss_models(self):
"""show_metrics() raises NotImplementedError for non-Nova models."""
trainer = self._make_trainer(latest_job="some-job")

def test_show_metrics_oss_without_mlflow_raises(self):
"""show_metrics() raises ValueError for non-Nova models without MLflow configured."""
trainer = self._make_trainer(latest_job=MagicMock(
training_job_name="some-job",
mlflow_config=None,
mlflow_details=None,
))
trainer._model_name = "test-oss-model"

with pytest.raises(NotImplementedError, match="only supported for Nova models"):
with pytest.raises(ValueError, match="requires MLflow to be configured"):
trainer.show_metrics()

@patch("sagemaker.train.base_trainer.plot_training_metrics")
def test_show_metrics_oss_with_mlflow_delegates(self, mock_plot):
"""show_metrics() for OSS models with MLflow configured calls plot_training_metrics."""
mock_job = MagicMock()
mock_job.training_job_name = "oss-sft-job"
mock_job.mlflow_config.mlflow_resource_arn = "arn:aws:sagemaker:us-east-1:012345678910:mlflow-app/app-123"
mock_job.mlflow_details.mlflow_run_id = "run-abc123"

trainer = self._make_trainer(latest_job=mock_job)
trainer._model_name = "test-oss-model"

trainer.show_metrics(metrics=["loss"])

mock_plot.assert_called_once_with(mock_job, metrics=["loss"])