Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions sagemaker-serve/src/sagemaker/serve/bedrock_model_builder.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -364,7 +364,7 @@ def deploy(
self._get_bedrock_client(), model_arn
)
if existing_deployment:
logger.warning(
logger.info(
"Reusing existing custom model %s and deployment %s "
"(matched model-source tag). No new resources were created. "
"Pass reuse_resources=False to force new resources.",
Expand All@@ -375,7 +375,7 @@ def deploy(
"modelArn": model_arn,
"customModelDeploymentArn": existing_deployment,
}
logger.warning(
logger.info(
"Reusing existing custom model %s (matched model-source tag); "
"creating a new deployment on it. Pass reuse_resources=False to "
"force a new model.",
Expand Down
4 changes: 2 additions & 2 deletions sagemaker-serve/src/sagemaker/serve/model_builder.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4061,7 +4061,7 @@ def build(
reusable_endpoint = self._find_reusable_endpoint()
if reusable_endpoint:
self._reused_endpoint_name = reusable_endpoint
logger.warning(
logger.info(
"Reusing existing Model %r (matched model-source tag). "
"No new Model will be created. Pass reuse_resources=False "
"to force a new Model.",
Expand DownExpand Up@@ -5528,7 +5528,7 @@ def deploy(
endpoint_name,
reusable_endpoint,
)
logger.warning(
logger.info(
"Reusing existing endpoint %r (matched model-source tag and "
"deployment configuration). No new resources were created. "
"Pass reuse_resources=False to force a new endpoint.",
Expand Down
50 changes: 36 additions & 14 deletions sagemaker-train/src/sagemaker/train/base_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -399,30 +399,40 @@ def show_metrics(
ValueError: If no training job has been run yet, no logs/metrics
are found, or MLflow is not configured for OSS models.
"""
# Validate that we have a training job to get metrics from
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics."
# Resolve the job reference. Prefer _latest_training_job (CreateTrainingJob),
# fall back to _latest_job (generic CreateJob API used by MTRL).
resolved_job = getattr(self, '_latest_training_job', None)
if resolved_job is None:
latest_job = getattr(self, '_latest_job', None)
if latest_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics. If training has already completed, set the "
"job name directly via trainer._latest_training_job = '<job-name>' or "
"trainer._latest_job = '<job-name>'."
)
resolved_job = (
latest_job.job_name if hasattr(latest_job, 'job_name') else str(latest_job)
)

# Route based on model type
model_name = getattr(self, '_model_name', None)
is_nova = _is_nova_model(model_name) if model_name else False

if is_nova:
return self._show_metrics_cloudwatch(metrics, starting_step, ending_step, start_time, end_time)
return self._show_metrics_cloudwatch(resolved_job, metrics, starting_step, ending_step, start_time, end_time)
else:
return self._show_metrics_mlflow(metrics, starting_step, ending_step)
return self._show_metrics_mlflow(resolved_job, metrics, starting_step, ending_step)

def _show_metrics_mlflow(
self,
resolved_job,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
) -> None:
"""Pull and plot training metrics from MLflow for non-Nova models."""
training_job = self._latest_training_job
training_job = resolved_job

# Resolve the TrainingJob object if it's a string
if isinstance(training_job, str):
Expand DownExpand Up@@ -456,6 +466,7 @@ def _show_metrics_mlflow(

def _show_metrics_cloudwatch(
self,
resolved_job,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
Expand All@@ -464,7 +475,7 @@ def _show_metrics_cloudwatch(
) -> Any:
"""Parse and plot training metrics from CloudWatch logs (Nova models)."""

training_job = self._latest_training_job
training_job = resolved_job
if hasattr(training_job, 'training_job_name'):
job_id = training_job.training_job_name
elif isinstance(training_job, str):
Expand DownExpand Up@@ -631,10 +642,21 @@ def stream_logs(self, poll: int = 5, start_time: Optional[Any] = None) -> None:
Raises:
ValueError: If no training job has been run yet.
"""
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train(wait=False) first, "
"then call .stream_logs() to stream logs in real-time."
# Resolve the job reference. Prefer _latest_training_job (CreateTrainingJob),
# fall back to _latest_job (generic CreateJob API used by MTRL).
resolved_job = getattr(self, '_latest_training_job', None)
if resolved_job is None:
latest_job = getattr(self, '_latest_job', None)
if latest_job is None:
raise ValueError(
"No training job found. Call .train(wait=False) first, "
"then call .stream_logs() to stream logs in real-time. "
"If training has already completed, set the job name directly via "
"trainer._latest_training_job = '<job-name>' or "
"trainer._latest_job = '<job-name>'."
)
resolved_job = (
latest_job.job_name if hasattr(latest_job, 'job_name') else str(latest_job)
)

# Resolve start_time for SMHP jobs
Expand All@@ -645,7 +667,7 @@ def stream_logs(self, poll: int = 5, start_time: Optional[Any] = None) -> None:
else:
start_time_ms = int(start_time)

training_job = self._latest_training_job
training_job = resolved_job
compute = getattr(self, 'compute', None)

if isinstance(compute, HyperPodCompute):
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -35,11 +35,13 @@
"SFT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"CPT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"RLVR": {"reward_score": SMTJ_RLVR_REWARD_SCORE_REGEX},
"MTRL": {"reward_score": SMTJ_RLVR_REWARD_SCORE_REGEX},
},
"smhp": {
"SFT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"CPT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"RLVR": {"reward_score": SMHP_RLVR_REWARD_SCORE_REGEX},
"MTRL": {"reward_score": SMHP_RLVR_REWARD_SCORE_REGEX},
},
}

Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -248,7 +248,7 @@ def is_multimodal_data(dataset: Union[str, "DataSet"]) -> bool:
True if multimodal fields detected, False otherwise
"""

logger.info(f"Auto-detecting whether dataset is multimodal: {dataset}")
logger.debug(f"Auto-detecting whether dataset is multimodal: {dataset}")

if isinstance(dataset, DataSet):
data_s3_path = dataset.source
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/cpt_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -40,7 +40,6 @@
from sagemaker.core.telemetry.constants import Feature

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class CPTTrainer(BaseTrainer):
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/dpo_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -31,7 +31,6 @@
from sagemaker.train.constants import get_sagemaker_hub_name

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class DPOTrainer(BaseTrainer):
Expand Down
2 changes: 2 additions & 0 deletions sagemaker-train/src/sagemaker/train/multi_turn_rl_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -173,6 +173,8 @@ class MultiTurnRLTrainer(BaseTrainer):
and 'job_name_prefix'. If not specified, no notifications are sent.
"""

_customization_technique = "MTRL"

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

What's the use of this variable?

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

It's for consistency. All the other trainers have it and will be called in BaseTrainer utils


def __init__(
self,
model: Union[str, ModelPackage],
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/sft_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -41,7 +41,6 @@
from sagemaker.core.training.constants import TrainingPlatform

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class SFTTrainer(BaseTrainer):
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions sagemaker-serve/src/sagemaker/serve/bedrock_model_builder.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -364,7 +364,7 @@ def deploy(
self._get_bedrock_client(), model_arn
)
if existing_deployment:
logger.warning(
logger.info(
"Reusing existing custom model %s and deployment %s "
"(matched model-source tag). No new resources were created. "
"Pass reuse_resources=False to force new resources.",
Expand All@@ -375,7 +375,7 @@ def deploy(
"modelArn": model_arn,
"customModelDeploymentArn": existing_deployment,
}
logger.warning(
logger.info(
"Reusing existing custom model %s (matched model-source tag); "
"creating a new deployment on it. Pass reuse_resources=False to "
"force a new model.",
Expand Down
4 changes: 2 additions & 2 deletions sagemaker-serve/src/sagemaker/serve/model_builder.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4061,7 +4061,7 @@ def build(
reusable_endpoint = self._find_reusable_endpoint()
if reusable_endpoint:
self._reused_endpoint_name = reusable_endpoint
logger.warning(
logger.info(
"Reusing existing Model %r (matched model-source tag). "
"No new Model will be created. Pass reuse_resources=False "
"to force a new Model.",
Expand DownExpand Up@@ -5528,7 +5528,7 @@ def deploy(
endpoint_name,
reusable_endpoint,
)
logger.warning(
logger.info(
"Reusing existing endpoint %r (matched model-source tag and "
"deployment configuration). No new resources were created. "
"Pass reuse_resources=False to force a new endpoint.",
Expand Down
50 changes: 36 additions & 14 deletions sagemaker-train/src/sagemaker/train/base_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -399,30 +399,40 @@ def show_metrics(
ValueError: If no training job has been run yet, no logs/metrics
are found, or MLflow is not configured for OSS models.
"""
# Validate that we have a training job to get metrics from
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics."
# Resolve the job reference. Prefer _latest_training_job (CreateTrainingJob),
# fall back to _latest_job (generic CreateJob API used by MTRL).
resolved_job = getattr(self, '_latest_training_job', None)
if resolved_job is None:
latest_job = getattr(self, '_latest_job', None)
if latest_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics. If training has already completed, set the "
"job name directly via trainer._latest_training_job = '<job-name>' or "
"trainer._latest_job = '<job-name>'."
)
resolved_job = (
latest_job.job_name if hasattr(latest_job, 'job_name') else str(latest_job)
)

# Route based on model type
model_name = getattr(self, '_model_name', None)
is_nova = _is_nova_model(model_name) if model_name else False

if is_nova:
return self._show_metrics_cloudwatch(metrics, starting_step, ending_step, start_time, end_time)
return self._show_metrics_cloudwatch(resolved_job, metrics, starting_step, ending_step, start_time, end_time)
else:
return self._show_metrics_mlflow(metrics, starting_step, ending_step)
return self._show_metrics_mlflow(resolved_job, metrics, starting_step, ending_step)

def _show_metrics_mlflow(
self,
resolved_job,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
) -> None:
"""Pull and plot training metrics from MLflow for non-Nova models."""
training_job = self._latest_training_job
training_job = resolved_job

# Resolve the TrainingJob object if it's a string
if isinstance(training_job, str):
Expand DownExpand Up@@ -456,6 +466,7 @@ def _show_metrics_mlflow(

def _show_metrics_cloudwatch(
self,
resolved_job,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
Expand All@@ -464,7 +475,7 @@ def _show_metrics_cloudwatch(
) -> Any:
"""Parse and plot training metrics from CloudWatch logs (Nova models)."""

training_job = self._latest_training_job
training_job = resolved_job
if hasattr(training_job, 'training_job_name'):
job_id = training_job.training_job_name
elif isinstance(training_job, str):
Expand DownExpand Up@@ -631,10 +642,21 @@ def stream_logs(self, poll: int = 5, start_time: Optional[Any] = None) -> None:
Raises:
ValueError: If no training job has been run yet.
"""
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train(wait=False) first, "
"then call .stream_logs() to stream logs in real-time."
# Resolve the job reference. Prefer _latest_training_job (CreateTrainingJob),
# fall back to _latest_job (generic CreateJob API used by MTRL).
resolved_job = getattr(self, '_latest_training_job', None)
if resolved_job is None:
latest_job = getattr(self, '_latest_job', None)
if latest_job is None:
raise ValueError(
"No training job found. Call .train(wait=False) first, "
"then call .stream_logs() to stream logs in real-time. "
"If training has already completed, set the job name directly via "
"trainer._latest_training_job = '<job-name>' or "
"trainer._latest_job = '<job-name>'."
)
resolved_job = (
latest_job.job_name if hasattr(latest_job, 'job_name') else str(latest_job)
)

# Resolve start_time for SMHP jobs
Expand All@@ -645,7 +667,7 @@ def stream_logs(self, poll: int = 5, start_time: Optional[Any] = None) -> None:
else:
start_time_ms = int(start_time)

training_job = self._latest_training_job
training_job = resolved_job
compute = getattr(self, 'compute', None)

if isinstance(compute, HyperPodCompute):
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -35,11 +35,13 @@
"SFT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"CPT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"RLVR": {"reward_score": SMTJ_RLVR_REWARD_SCORE_REGEX},
"MTRL": {"reward_score": SMTJ_RLVR_REWARD_SCORE_REGEX},
},
"smhp": {
"SFT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"CPT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"RLVR": {"reward_score": SMHP_RLVR_REWARD_SCORE_REGEX},
"MTRL": {"reward_score": SMHP_RLVR_REWARD_SCORE_REGEX},
},
}

Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -248,7 +248,7 @@ def is_multimodal_data(dataset: Union[str, "DataSet"]) -> bool:
True if multimodal fields detected, False otherwise
"""

logger.info(f"Auto-detecting whether dataset is multimodal: {dataset}")
logger.debug(f"Auto-detecting whether dataset is multimodal: {dataset}")

if isinstance(dataset, DataSet):
data_s3_path = dataset.source
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/cpt_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -40,7 +40,6 @@
from sagemaker.core.telemetry.constants import Feature

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class CPTTrainer(BaseTrainer):
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/dpo_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -31,7 +31,6 @@
from sagemaker.train.constants import get_sagemaker_hub_name

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class DPOTrainer(BaseTrainer):
Expand Down
2 changes: 2 additions & 0 deletions sagemaker-train/src/sagemaker/train/multi_turn_rl_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -173,6 +173,8 @@ class MultiTurnRLTrainer(BaseTrainer):
and 'job_name_prefix'. If not specified, no notifications are sent.
"""

_customization_technique = "MTRL"

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

What's the use of this variable?

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

It's for consistency. All the other trainers have it and will be called in BaseTrainer utils


def __init__(
self,
model: Union[str, ModelPackage],
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/sft_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -41,7 +41,6 @@
from sagemaker.core.training.constants import TrainingPlatform

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class SFTTrainer(BaseTrainer):
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions sagemaker-serve/src/sagemaker/serve/bedrock_model_builder.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -364,7 +364,7 @@ def deploy(
self._get_bedrock_client(), model_arn
)
if existing_deployment:
logger.warning(
logger.info(
"Reusing existing custom model %s and deployment %s "
"(matched model-source tag). No new resources were created. "
"Pass reuse_resources=False to force new resources.",
Expand All@@ -375,7 +375,7 @@ def deploy(
"modelArn": model_arn,
"customModelDeploymentArn": existing_deployment,
}
logger.warning(
logger.info(
"Reusing existing custom model %s (matched model-source tag); "
"creating a new deployment on it. Pass reuse_resources=False to "
"force a new model.",
Expand Down
4 changes: 2 additions & 2 deletions sagemaker-serve/src/sagemaker/serve/model_builder.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4061,7 +4061,7 @@ def build(
reusable_endpoint = self._find_reusable_endpoint()
if reusable_endpoint:
self._reused_endpoint_name = reusable_endpoint
logger.warning(
logger.info(
"Reusing existing Model %r (matched model-source tag). "
"No new Model will be created. Pass reuse_resources=False "
"to force a new Model.",
Expand DownExpand Up@@ -5528,7 +5528,7 @@ def deploy(
endpoint_name,
reusable_endpoint,
)
logger.warning(
logger.info(
"Reusing existing endpoint %r (matched model-source tag and "
"deployment configuration). No new resources were created. "
"Pass reuse_resources=False to force a new endpoint.",
Expand Down
50 changes: 36 additions & 14 deletions sagemaker-train/src/sagemaker/train/base_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -399,30 +399,40 @@ def show_metrics(
ValueError: If no training job has been run yet, no logs/metrics
are found, or MLflow is not configured for OSS models.
"""
# Validate that we have a training job to get metrics from
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics."
# Resolve the job reference. Prefer _latest_training_job (CreateTrainingJob),
# fall back to _latest_job (generic CreateJob API used by MTRL).
resolved_job = getattr(self, '_latest_training_job', None)
if resolved_job is None:
latest_job = getattr(self, '_latest_job', None)
if latest_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics. If training has already completed, set the "
"job name directly via trainer._latest_training_job = '<job-name>' or "
"trainer._latest_job = '<job-name>'."
)
resolved_job = (
latest_job.job_name if hasattr(latest_job, 'job_name') else str(latest_job)
)

# Route based on model type
model_name = getattr(self, '_model_name', None)
is_nova = _is_nova_model(model_name) if model_name else False

if is_nova:
return self._show_metrics_cloudwatch(metrics, starting_step, ending_step, start_time, end_time)
return self._show_metrics_cloudwatch(resolved_job, metrics, starting_step, ending_step, start_time, end_time)
else:
return self._show_metrics_mlflow(metrics, starting_step, ending_step)
return self._show_metrics_mlflow(resolved_job, metrics, starting_step, ending_step)

def _show_metrics_mlflow(
self,
resolved_job,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
) -> None:
"""Pull and plot training metrics from MLflow for non-Nova models."""
training_job = self._latest_training_job
training_job = resolved_job

# Resolve the TrainingJob object if it's a string
if isinstance(training_job, str):
Expand DownExpand Up@@ -456,6 +466,7 @@ def _show_metrics_mlflow(

def _show_metrics_cloudwatch(
self,
resolved_job,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
Expand All@@ -464,7 +475,7 @@ def _show_metrics_cloudwatch(
) -> Any:
"""Parse and plot training metrics from CloudWatch logs (Nova models)."""

training_job = self._latest_training_job
training_job = resolved_job
if hasattr(training_job, 'training_job_name'):
job_id = training_job.training_job_name
elif isinstance(training_job, str):
Expand DownExpand Up@@ -631,10 +642,21 @@ def stream_logs(self, poll: int = 5, start_time: Optional[Any] = None) -> None:
Raises:
ValueError: If no training job has been run yet.
"""
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train(wait=False) first, "
"then call .stream_logs() to stream logs in real-time."
# Resolve the job reference. Prefer _latest_training_job (CreateTrainingJob),
# fall back to _latest_job (generic CreateJob API used by MTRL).
resolved_job = getattr(self, '_latest_training_job', None)
if resolved_job is None:
latest_job = getattr(self, '_latest_job', None)
if latest_job is None:
raise ValueError(
"No training job found. Call .train(wait=False) first, "
"then call .stream_logs() to stream logs in real-time. "
"If training has already completed, set the job name directly via "
"trainer._latest_training_job = '<job-name>' or "
"trainer._latest_job = '<job-name>'."
)
resolved_job = (
latest_job.job_name if hasattr(latest_job, 'job_name') else str(latest_job)
)

# Resolve start_time for SMHP jobs
Expand All@@ -645,7 +667,7 @@ def stream_logs(self, poll: int = 5, start_time: Optional[Any] = None) -> None:
else:
start_time_ms = int(start_time)

training_job = self._latest_training_job
training_job = resolved_job
compute = getattr(self, 'compute', None)

if isinstance(compute, HyperPodCompute):
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -35,11 +35,13 @@
"SFT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"CPT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"RLVR": {"reward_score": SMTJ_RLVR_REWARD_SCORE_REGEX},
"MTRL": {"reward_score": SMTJ_RLVR_REWARD_SCORE_REGEX},
},
"smhp": {
"SFT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"CPT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"RLVR": {"reward_score": SMHP_RLVR_REWARD_SCORE_REGEX},
"MTRL": {"reward_score": SMHP_RLVR_REWARD_SCORE_REGEX},
},
}

Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -248,7 +248,7 @@ def is_multimodal_data(dataset: Union[str, "DataSet"]) -> bool:
True if multimodal fields detected, False otherwise
"""

logger.info(f"Auto-detecting whether dataset is multimodal: {dataset}")
logger.debug(f"Auto-detecting whether dataset is multimodal: {dataset}")

if isinstance(dataset, DataSet):
data_s3_path = dataset.source
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/cpt_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -40,7 +40,6 @@
from sagemaker.core.telemetry.constants import Feature

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class CPTTrainer(BaseTrainer):
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/dpo_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -31,7 +31,6 @@
from sagemaker.train.constants import get_sagemaker_hub_name

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class DPOTrainer(BaseTrainer):
Expand Down
2 changes: 2 additions & 0 deletions sagemaker-train/src/sagemaker/train/multi_turn_rl_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -173,6 +173,8 @@ class MultiTurnRLTrainer(BaseTrainer):
and 'job_name_prefix'. If not specified, no notifications are sent.
"""

_customization_technique = "MTRL"

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

What's the use of this variable?

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

It's for consistency. All the other trainers have it and will be called in BaseTrainer utils


def __init__(
self,
model: Union[str, ModelPackage],
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/sft_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -41,7 +41,6 @@
from sagemaker.core.training.constants import TrainingPlatform

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class SFTTrainer(BaseTrainer):
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions sagemaker-serve/src/sagemaker/serve/bedrock_model_builder.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -364,7 +364,7 @@ def deploy(
self._get_bedrock_client(), model_arn
)
if existing_deployment:
logger.warning(
logger.info(
"Reusing existing custom model %s and deployment %s "
"(matched model-source tag). No new resources were created. "
"Pass reuse_resources=False to force new resources.",
Expand All@@ -375,7 +375,7 @@ def deploy(
"modelArn": model_arn,
"customModelDeploymentArn": existing_deployment,
}
logger.warning(
logger.info(
"Reusing existing custom model %s (matched model-source tag); "
"creating a new deployment on it. Pass reuse_resources=False to "
"force a new model.",
Expand Down
4 changes: 2 additions & 2 deletions sagemaker-serve/src/sagemaker/serve/model_builder.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4061,7 +4061,7 @@ def build(
reusable_endpoint = self._find_reusable_endpoint()
if reusable_endpoint:
self._reused_endpoint_name = reusable_endpoint
logger.warning(
logger.info(
"Reusing existing Model %r (matched model-source tag). "
"No new Model will be created. Pass reuse_resources=False "
"to force a new Model.",
Expand DownExpand Up@@ -5528,7 +5528,7 @@ def deploy(
endpoint_name,
reusable_endpoint,
)
logger.warning(
logger.info(
"Reusing existing endpoint %r (matched model-source tag and "
"deployment configuration). No new resources were created. "
"Pass reuse_resources=False to force a new endpoint.",
Expand Down
50 changes: 36 additions & 14 deletions sagemaker-train/src/sagemaker/train/base_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -399,30 +399,40 @@ def show_metrics(
ValueError: If no training job has been run yet, no logs/metrics
are found, or MLflow is not configured for OSS models.
"""
# Validate that we have a training job to get metrics from
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics."
# Resolve the job reference. Prefer _latest_training_job (CreateTrainingJob),
# fall back to _latest_job (generic CreateJob API used by MTRL).
resolved_job = getattr(self, '_latest_training_job', None)
if resolved_job is None:
latest_job = getattr(self, '_latest_job', None)
if latest_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics. If training has already completed, set the "
"job name directly via trainer._latest_training_job = '<job-name>' or "
"trainer._latest_job = '<job-name>'."
)
resolved_job = (
latest_job.job_name if hasattr(latest_job, 'job_name') else str(latest_job)
)

# Route based on model type
model_name = getattr(self, '_model_name', None)
is_nova = _is_nova_model(model_name) if model_name else False

if is_nova:
return self._show_metrics_cloudwatch(metrics, starting_step, ending_step, start_time, end_time)
return self._show_metrics_cloudwatch(resolved_job, metrics, starting_step, ending_step, start_time, end_time)
else:
return self._show_metrics_mlflow(metrics, starting_step, ending_step)
return self._show_metrics_mlflow(resolved_job, metrics, starting_step, ending_step)

def _show_metrics_mlflow(
self,
resolved_job,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
) -> None:
"""Pull and plot training metrics from MLflow for non-Nova models."""
training_job = self._latest_training_job
training_job = resolved_job

# Resolve the TrainingJob object if it's a string
if isinstance(training_job, str):
Expand DownExpand Up@@ -456,6 +466,7 @@ def _show_metrics_mlflow(

def _show_metrics_cloudwatch(
self,
resolved_job,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
Expand All@@ -464,7 +475,7 @@ def _show_metrics_cloudwatch(
) -> Any:
"""Parse and plot training metrics from CloudWatch logs (Nova models)."""

training_job = self._latest_training_job
training_job = resolved_job
if hasattr(training_job, 'training_job_name'):
job_id = training_job.training_job_name
elif isinstance(training_job, str):
Expand DownExpand Up@@ -631,10 +642,21 @@ def stream_logs(self, poll: int = 5, start_time: Optional[Any] = None) -> None:
Raises:
ValueError: If no training job has been run yet.
"""
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train(wait=False) first, "
"then call .stream_logs() to stream logs in real-time."
# Resolve the job reference. Prefer _latest_training_job (CreateTrainingJob),
# fall back to _latest_job (generic CreateJob API used by MTRL).
resolved_job = getattr(self, '_latest_training_job', None)
if resolved_job is None:
latest_job = getattr(self, '_latest_job', None)
if latest_job is None:
raise ValueError(
"No training job found. Call .train(wait=False) first, "
"then call .stream_logs() to stream logs in real-time. "
"If training has already completed, set the job name directly via "
"trainer._latest_training_job = '<job-name>' or "
"trainer._latest_job = '<job-name>'."
)
resolved_job = (
latest_job.job_name if hasattr(latest_job, 'job_name') else str(latest_job)
)

# Resolve start_time for SMHP jobs
Expand All@@ -645,7 +667,7 @@ def stream_logs(self, poll: int = 5, start_time: Optional[Any] = None) -> None:
else:
start_time_ms = int(start_time)

training_job = self._latest_training_job
training_job = resolved_job
compute = getattr(self, 'compute', None)

if isinstance(compute, HyperPodCompute):
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -35,11 +35,13 @@
"SFT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"CPT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"RLVR": {"reward_score": SMTJ_RLVR_REWARD_SCORE_REGEX},
"MTRL": {"reward_score": SMTJ_RLVR_REWARD_SCORE_REGEX},
},
"smhp": {
"SFT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"CPT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"RLVR": {"reward_score": SMHP_RLVR_REWARD_SCORE_REGEX},
"MTRL": {"reward_score": SMHP_RLVR_REWARD_SCORE_REGEX},
},
}

Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -248,7 +248,7 @@ def is_multimodal_data(dataset: Union[str, "DataSet"]) -> bool:
True if multimodal fields detected, False otherwise
"""

logger.info(f"Auto-detecting whether dataset is multimodal: {dataset}")
logger.debug(f"Auto-detecting whether dataset is multimodal: {dataset}")

if isinstance(dataset, DataSet):
data_s3_path = dataset.source
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/cpt_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -40,7 +40,6 @@
from sagemaker.core.telemetry.constants import Feature

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class CPTTrainer(BaseTrainer):
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/dpo_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -31,7 +31,6 @@
from sagemaker.train.constants import get_sagemaker_hub_name

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class DPOTrainer(BaseTrainer):
Expand Down
2 changes: 2 additions & 0 deletions sagemaker-train/src/sagemaker/train/multi_turn_rl_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -173,6 +173,8 @@ class MultiTurnRLTrainer(BaseTrainer):
and 'job_name_prefix'. If not specified, no notifications are sent.
"""

_customization_technique = "MTRL"

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

What's the use of this variable?

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

It's for consistency. All the other trainers have it and will be called in BaseTrainer utils


def __init__(
self,
model: Union[str, ModelPackage],
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/sft_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -41,7 +41,6 @@
from sagemaker.core.training.constants import TrainingPlatform

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class SFTTrainer(BaseTrainer):
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions sagemaker-serve/src/sagemaker/serve/bedrock_model_builder.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -364,7 +364,7 @@ def deploy(
self._get_bedrock_client(), model_arn
)
if existing_deployment:
logger.warning(
logger.info(
"Reusing existing custom model %s and deployment %s "
"(matched model-source tag). No new resources were created. "
"Pass reuse_resources=False to force new resources.",
Expand All@@ -375,7 +375,7 @@ def deploy(
"modelArn": model_arn,
"customModelDeploymentArn": existing_deployment,
}
logger.warning(
logger.info(
"Reusing existing custom model %s (matched model-source tag); "
"creating a new deployment on it. Pass reuse_resources=False to "
"force a new model.",
Expand Down
4 changes: 2 additions & 2 deletions sagemaker-serve/src/sagemaker/serve/model_builder.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4061,7 +4061,7 @@ def build(
reusable_endpoint = self._find_reusable_endpoint()
if reusable_endpoint:
self._reused_endpoint_name = reusable_endpoint
logger.warning(
logger.info(
"Reusing existing Model %r (matched model-source tag). "
"No new Model will be created. Pass reuse_resources=False "
"to force a new Model.",
Expand DownExpand Up@@ -5528,7 +5528,7 @@ def deploy(
endpoint_name,
reusable_endpoint,
)
logger.warning(
logger.info(
"Reusing existing endpoint %r (matched model-source tag and "
"deployment configuration). No new resources were created. "
"Pass reuse_resources=False to force a new endpoint.",
Expand Down
50 changes: 36 additions & 14 deletions sagemaker-train/src/sagemaker/train/base_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -399,30 +399,40 @@ def show_metrics(
ValueError: If no training job has been run yet, no logs/metrics
are found, or MLflow is not configured for OSS models.
"""
# Validate that we have a training job to get metrics from
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics."
# Resolve the job reference. Prefer _latest_training_job (CreateTrainingJob),
# fall back to _latest_job (generic CreateJob API used by MTRL).
resolved_job = getattr(self, '_latest_training_job', None)
if resolved_job is None:
latest_job = getattr(self, '_latest_job', None)
if latest_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics. If training has already completed, set the "
"job name directly via trainer._latest_training_job = '<job-name>' or "
"trainer._latest_job = '<job-name>'."
)
resolved_job = (
latest_job.job_name if hasattr(latest_job, 'job_name') else str(latest_job)
)

# Route based on model type
model_name = getattr(self, '_model_name', None)
is_nova = _is_nova_model(model_name) if model_name else False

if is_nova:
return self._show_metrics_cloudwatch(metrics, starting_step, ending_step, start_time, end_time)
return self._show_metrics_cloudwatch(resolved_job, metrics, starting_step, ending_step, start_time, end_time)
else:
return self._show_metrics_mlflow(metrics, starting_step, ending_step)
return self._show_metrics_mlflow(resolved_job, metrics, starting_step, ending_step)

def _show_metrics_mlflow(
self,
resolved_job,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
) -> None:
"""Pull and plot training metrics from MLflow for non-Nova models."""
training_job = self._latest_training_job
training_job = resolved_job

# Resolve the TrainingJob object if it's a string
if isinstance(training_job, str):
Expand DownExpand Up@@ -456,6 +466,7 @@ def _show_metrics_mlflow(

def _show_metrics_cloudwatch(
self,
resolved_job,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
Expand All@@ -464,7 +475,7 @@ def _show_metrics_cloudwatch(
) -> Any:
"""Parse and plot training metrics from CloudWatch logs (Nova models)."""

training_job = self._latest_training_job
training_job = resolved_job
if hasattr(training_job, 'training_job_name'):
job_id = training_job.training_job_name
elif isinstance(training_job, str):
Expand DownExpand Up@@ -631,10 +642,21 @@ def stream_logs(self, poll: int = 5, start_time: Optional[Any] = None) -> None:
Raises:
ValueError: If no training job has been run yet.
"""
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train(wait=False) first, "
"then call .stream_logs() to stream logs in real-time."
# Resolve the job reference. Prefer _latest_training_job (CreateTrainingJob),
# fall back to _latest_job (generic CreateJob API used by MTRL).
resolved_job = getattr(self, '_latest_training_job', None)
if resolved_job is None:
latest_job = getattr(self, '_latest_job', None)
if latest_job is None:
raise ValueError(
"No training job found. Call .train(wait=False) first, "
"then call .stream_logs() to stream logs in real-time. "
"If training has already completed, set the job name directly via "
"trainer._latest_training_job = '<job-name>' or "
"trainer._latest_job = '<job-name>'."
)
resolved_job = (
latest_job.job_name if hasattr(latest_job, 'job_name') else str(latest_job)
)

# Resolve start_time for SMHP jobs
Expand All@@ -645,7 +667,7 @@ def stream_logs(self, poll: int = 5, start_time: Optional[Any] = None) -> None:
else:
start_time_ms = int(start_time)

training_job = self._latest_training_job
training_job = resolved_job
compute = getattr(self, 'compute', None)

if isinstance(compute, HyperPodCompute):
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -35,11 +35,13 @@
"SFT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"CPT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"RLVR": {"reward_score": SMTJ_RLVR_REWARD_SCORE_REGEX},
"MTRL": {"reward_score": SMTJ_RLVR_REWARD_SCORE_REGEX},
},
"smhp": {
"SFT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"CPT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"RLVR": {"reward_score": SMHP_RLVR_REWARD_SCORE_REGEX},
"MTRL": {"reward_score": SMHP_RLVR_REWARD_SCORE_REGEX},
},
}

Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -248,7 +248,7 @@ def is_multimodal_data(dataset: Union[str, "DataSet"]) -> bool:
True if multimodal fields detected, False otherwise
"""

logger.info(f"Auto-detecting whether dataset is multimodal: {dataset}")
logger.debug(f"Auto-detecting whether dataset is multimodal: {dataset}")

if isinstance(dataset, DataSet):
data_s3_path = dataset.source
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/cpt_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -40,7 +40,6 @@
from sagemaker.core.telemetry.constants import Feature

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class CPTTrainer(BaseTrainer):
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/dpo_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -31,7 +31,6 @@
from sagemaker.train.constants import get_sagemaker_hub_name

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class DPOTrainer(BaseTrainer):
Expand Down
2 changes: 2 additions & 0 deletions sagemaker-train/src/sagemaker/train/multi_turn_rl_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -173,6 +173,8 @@ class MultiTurnRLTrainer(BaseTrainer):
and 'job_name_prefix'. If not specified, no notifications are sent.
"""

_customization_technique = "MTRL"

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

What's the use of this variable?

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

It's for consistency. All the other trainers have it and will be called in BaseTrainer utils


def __init__(
self,
model: Union[str, ModelPackage],
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/sft_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -41,7 +41,6 @@
from sagemaker.core.training.constants import TrainingPlatform

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class SFTTrainer(BaseTrainer):
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions sagemaker-serve/src/sagemaker/serve/bedrock_model_builder.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -364,7 +364,7 @@ def deploy(
self._get_bedrock_client(), model_arn
)
if existing_deployment:
logger.warning(
logger.info(
"Reusing existing custom model %s and deployment %s "
"(matched model-source tag). No new resources were created. "
"Pass reuse_resources=False to force new resources.",
Expand All@@ -375,7 +375,7 @@ def deploy(
"modelArn": model_arn,
"customModelDeploymentArn": existing_deployment,
}
logger.warning(
logger.info(
"Reusing existing custom model %s (matched model-source tag); "
"creating a new deployment on it. Pass reuse_resources=False to "
"force a new model.",
Expand Down
4 changes: 2 additions & 2 deletions sagemaker-serve/src/sagemaker/serve/model_builder.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4061,7 +4061,7 @@ def build(
reusable_endpoint = self._find_reusable_endpoint()
if reusable_endpoint:
self._reused_endpoint_name = reusable_endpoint
logger.warning(
logger.info(
"Reusing existing Model %r (matched model-source tag). "
"No new Model will be created. Pass reuse_resources=False "
"to force a new Model.",
Expand DownExpand Up@@ -5528,7 +5528,7 @@ def deploy(
endpoint_name,
reusable_endpoint,
)
logger.warning(
logger.info(
"Reusing existing endpoint %r (matched model-source tag and "
"deployment configuration). No new resources were created. "
"Pass reuse_resources=False to force a new endpoint.",
Expand Down
50 changes: 36 additions & 14 deletions sagemaker-train/src/sagemaker/train/base_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -399,30 +399,40 @@ def show_metrics(
ValueError: If no training job has been run yet, no logs/metrics
are found, or MLflow is not configured for OSS models.
"""
# Validate that we have a training job to get metrics from
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics."
# Resolve the job reference. Prefer _latest_training_job (CreateTrainingJob),
# fall back to _latest_job (generic CreateJob API used by MTRL).
resolved_job = getattr(self, '_latest_training_job', None)
if resolved_job is None:
latest_job = getattr(self, '_latest_job', None)
if latest_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics. If training has already completed, set the "
"job name directly via trainer._latest_training_job = '<job-name>' or "
"trainer._latest_job = '<job-name>'."
)
resolved_job = (
latest_job.job_name if hasattr(latest_job, 'job_name') else str(latest_job)
)

# Route based on model type
model_name = getattr(self, '_model_name', None)
is_nova = _is_nova_model(model_name) if model_name else False

if is_nova:
return self._show_metrics_cloudwatch(metrics, starting_step, ending_step, start_time, end_time)
return self._show_metrics_cloudwatch(resolved_job, metrics, starting_step, ending_step, start_time, end_time)
else:
return self._show_metrics_mlflow(metrics, starting_step, ending_step)
return self._show_metrics_mlflow(resolved_job, metrics, starting_step, ending_step)

def _show_metrics_mlflow(
self,
resolved_job,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
) -> None:
"""Pull and plot training metrics from MLflow for non-Nova models."""
training_job = self._latest_training_job
training_job = resolved_job

# Resolve the TrainingJob object if it's a string
if isinstance(training_job, str):
Expand DownExpand Up@@ -456,6 +466,7 @@ def _show_metrics_mlflow(

def _show_metrics_cloudwatch(
self,
resolved_job,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
Expand All@@ -464,7 +475,7 @@ def _show_metrics_cloudwatch(
) -> Any:
"""Parse and plot training metrics from CloudWatch logs (Nova models)."""

training_job = self._latest_training_job
training_job = resolved_job
if hasattr(training_job, 'training_job_name'):
job_id = training_job.training_job_name
elif isinstance(training_job, str):
Expand DownExpand Up@@ -631,10 +642,21 @@ def stream_logs(self, poll: int = 5, start_time: Optional[Any] = None) -> None:
Raises:
ValueError: If no training job has been run yet.
"""
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train(wait=False) first, "
"then call .stream_logs() to stream logs in real-time."
# Resolve the job reference. Prefer _latest_training_job (CreateTrainingJob),
# fall back to _latest_job (generic CreateJob API used by MTRL).
resolved_job = getattr(self, '_latest_training_job', None)
if resolved_job is None:
latest_job = getattr(self, '_latest_job', None)
if latest_job is None:
raise ValueError(
"No training job found. Call .train(wait=False) first, "
"then call .stream_logs() to stream logs in real-time. "
"If training has already completed, set the job name directly via "
"trainer._latest_training_job = '<job-name>' or "
"trainer._latest_job = '<job-name>'."
)
resolved_job = (
latest_job.job_name if hasattr(latest_job, 'job_name') else str(latest_job)
)

# Resolve start_time for SMHP jobs
Expand All@@ -645,7 +667,7 @@ def stream_logs(self, poll: int = 5, start_time: Optional[Any] = None) -> None:
else:
start_time_ms = int(start_time)

training_job = self._latest_training_job
training_job = resolved_job
compute = getattr(self, 'compute', None)

if isinstance(compute, HyperPodCompute):
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -35,11 +35,13 @@
"SFT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"CPT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"RLVR": {"reward_score": SMTJ_RLVR_REWARD_SCORE_REGEX},
"MTRL": {"reward_score": SMTJ_RLVR_REWARD_SCORE_REGEX},
},
"smhp": {
"SFT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"CPT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"RLVR": {"reward_score": SMHP_RLVR_REWARD_SCORE_REGEX},
"MTRL": {"reward_score": SMHP_RLVR_REWARD_SCORE_REGEX},
},
}

Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -248,7 +248,7 @@ def is_multimodal_data(dataset: Union[str, "DataSet"]) -> bool:
True if multimodal fields detected, False otherwise
"""

logger.info(f"Auto-detecting whether dataset is multimodal: {dataset}")
logger.debug(f"Auto-detecting whether dataset is multimodal: {dataset}")

if isinstance(dataset, DataSet):
data_s3_path = dataset.source
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/cpt_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -40,7 +40,6 @@
from sagemaker.core.telemetry.constants import Feature

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class CPTTrainer(BaseTrainer):
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/dpo_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -31,7 +31,6 @@
from sagemaker.train.constants import get_sagemaker_hub_name

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class DPOTrainer(BaseTrainer):
Expand Down
2 changes: 2 additions & 0 deletions sagemaker-train/src/sagemaker/train/multi_turn_rl_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -173,6 +173,8 @@ class MultiTurnRLTrainer(BaseTrainer):
and 'job_name_prefix'. If not specified, no notifications are sent.
"""

_customization_technique = "MTRL"

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

What's the use of this variable?

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

It's for consistency. All the other trainers have it and will be called in BaseTrainer utils


def __init__(
self,
model: Union[str, ModelPackage],
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/sft_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -41,7 +41,6 @@
from sagemaker.core.training.constants import TrainingPlatform

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class SFTTrainer(BaseTrainer):
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions sagemaker-serve/src/sagemaker/serve/bedrock_model_builder.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -364,7 +364,7 @@ def deploy(
self._get_bedrock_client(), model_arn
)
if existing_deployment:
logger.warning(
logger.info(
"Reusing existing custom model %s and deployment %s "
"(matched model-source tag). No new resources were created. "
"Pass reuse_resources=False to force new resources.",
Expand All@@ -375,7 +375,7 @@ def deploy(
"modelArn": model_arn,
"customModelDeploymentArn": existing_deployment,
}
logger.warning(
logger.info(
"Reusing existing custom model %s (matched model-source tag); "
"creating a new deployment on it. Pass reuse_resources=False to "
"force a new model.",
Expand Down
4 changes: 2 additions & 2 deletions sagemaker-serve/src/sagemaker/serve/model_builder.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4061,7 +4061,7 @@ def build(
reusable_endpoint = self._find_reusable_endpoint()
if reusable_endpoint:
self._reused_endpoint_name = reusable_endpoint
logger.warning(
logger.info(
"Reusing existing Model %r (matched model-source tag). "
"No new Model will be created. Pass reuse_resources=False "
"to force a new Model.",
Expand DownExpand Up@@ -5528,7 +5528,7 @@ def deploy(
endpoint_name,
reusable_endpoint,
)
logger.warning(
logger.info(
"Reusing existing endpoint %r (matched model-source tag and "
"deployment configuration). No new resources were created. "
"Pass reuse_resources=False to force a new endpoint.",
Expand Down
50 changes: 36 additions & 14 deletions sagemaker-train/src/sagemaker/train/base_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -399,30 +399,40 @@ def show_metrics(
ValueError: If no training job has been run yet, no logs/metrics
are found, or MLflow is not configured for OSS models.
"""
# Validate that we have a training job to get metrics from
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics."
# Resolve the job reference. Prefer _latest_training_job (CreateTrainingJob),
# fall back to _latest_job (generic CreateJob API used by MTRL).
resolved_job = getattr(self, '_latest_training_job', None)
if resolved_job is None:
latest_job = getattr(self, '_latest_job', None)
if latest_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics. If training has already completed, set the "
"job name directly via trainer._latest_training_job = '<job-name>' or "
"trainer._latest_job = '<job-name>'."
)
resolved_job = (
latest_job.job_name if hasattr(latest_job, 'job_name') else str(latest_job)
)

# Route based on model type
model_name = getattr(self, '_model_name', None)
is_nova = _is_nova_model(model_name) if model_name else False

if is_nova:
return self._show_metrics_cloudwatch(metrics, starting_step, ending_step, start_time, end_time)
return self._show_metrics_cloudwatch(resolved_job, metrics, starting_step, ending_step, start_time, end_time)
else:
return self._show_metrics_mlflow(metrics, starting_step, ending_step)
return self._show_metrics_mlflow(resolved_job, metrics, starting_step, ending_step)

def _show_metrics_mlflow(
self,
resolved_job,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
) -> None:
"""Pull and plot training metrics from MLflow for non-Nova models."""
training_job = self._latest_training_job
training_job = resolved_job

# Resolve the TrainingJob object if it's a string
if isinstance(training_job, str):
Expand DownExpand Up@@ -456,6 +466,7 @@ def _show_metrics_mlflow(

def _show_metrics_cloudwatch(
self,
resolved_job,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
Expand All@@ -464,7 +475,7 @@ def _show_metrics_cloudwatch(
) -> Any:
"""Parse and plot training metrics from CloudWatch logs (Nova models)."""

training_job = self._latest_training_job
training_job = resolved_job
if hasattr(training_job, 'training_job_name'):
job_id = training_job.training_job_name
elif isinstance(training_job, str):
Expand DownExpand Up@@ -631,10 +642,21 @@ def stream_logs(self, poll: int = 5, start_time: Optional[Any] = None) -> None:
Raises:
ValueError: If no training job has been run yet.
"""
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train(wait=False) first, "
"then call .stream_logs() to stream logs in real-time."
# Resolve the job reference. Prefer _latest_training_job (CreateTrainingJob),
# fall back to _latest_job (generic CreateJob API used by MTRL).
resolved_job = getattr(self, '_latest_training_job', None)
if resolved_job is None:
latest_job = getattr(self, '_latest_job', None)
if latest_job is None:
raise ValueError(
"No training job found. Call .train(wait=False) first, "
"then call .stream_logs() to stream logs in real-time. "
"If training has already completed, set the job name directly via "
"trainer._latest_training_job = '<job-name>' or "
"trainer._latest_job = '<job-name>'."
)
resolved_job = (
latest_job.job_name if hasattr(latest_job, 'job_name') else str(latest_job)
)

# Resolve start_time for SMHP jobs
Expand All@@ -645,7 +667,7 @@ def stream_logs(self, poll: int = 5, start_time: Optional[Any] = None) -> None:
else:
start_time_ms = int(start_time)

training_job = self._latest_training_job
training_job = resolved_job
compute = getattr(self, 'compute', None)

if isinstance(compute, HyperPodCompute):
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -35,11 +35,13 @@
"SFT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"CPT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"RLVR": {"reward_score": SMTJ_RLVR_REWARD_SCORE_REGEX},
"MTRL": {"reward_score": SMTJ_RLVR_REWARD_SCORE_REGEX},
},
"smhp": {
"SFT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"CPT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"RLVR": {"reward_score": SMHP_RLVR_REWARD_SCORE_REGEX},
"MTRL": {"reward_score": SMHP_RLVR_REWARD_SCORE_REGEX},
},
}

Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -248,7 +248,7 @@ def is_multimodal_data(dataset: Union[str, "DataSet"]) -> bool:
True if multimodal fields detected, False otherwise
"""

logger.info(f"Auto-detecting whether dataset is multimodal: {dataset}")
logger.debug(f"Auto-detecting whether dataset is multimodal: {dataset}")

if isinstance(dataset, DataSet):
data_s3_path = dataset.source
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/cpt_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -40,7 +40,6 @@
from sagemaker.core.telemetry.constants import Feature

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class CPTTrainer(BaseTrainer):
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/dpo_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -31,7 +31,6 @@
from sagemaker.train.constants import get_sagemaker_hub_name

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class DPOTrainer(BaseTrainer):
Expand Down
2 changes: 2 additions & 0 deletions sagemaker-train/src/sagemaker/train/multi_turn_rl_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -173,6 +173,8 @@ class MultiTurnRLTrainer(BaseTrainer):
and 'job_name_prefix'. If not specified, no notifications are sent.
"""

_customization_technique = "MTRL"

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

What's the use of this variable?

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

It's for consistency. All the other trainers have it and will be called in BaseTrainer utils


def __init__(
self,
model: Union[str, ModelPackage],
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/sft_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -41,7 +41,6 @@
from sagemaker.core.training.constants import TrainingPlatform

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class SFTTrainer(BaseTrainer):
Expand Down
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions sagemaker-serve/src/sagemaker/serve/bedrock_model_builder.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -364,7 +364,7 @@ def deploy(
self._get_bedrock_client(), model_arn
)
if existing_deployment:
logger.warning(
logger.info(
"Reusing existing custom model %s and deployment %s "
"(matched model-source tag). No new resources were created. "
"Pass reuse_resources=False to force new resources.",
Expand All@@ -375,7 +375,7 @@ def deploy(
"modelArn": model_arn,
"customModelDeploymentArn": existing_deployment,
}
logger.warning(
logger.info(
"Reusing existing custom model %s (matched model-source tag); "
"creating a new deployment on it. Pass reuse_resources=False to "
"force a new model.",
Expand Down
4 changes: 2 additions & 2 deletions sagemaker-serve/src/sagemaker/serve/model_builder.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4061,7 +4061,7 @@ def build(
reusable_endpoint = self._find_reusable_endpoint()
if reusable_endpoint:
self._reused_endpoint_name = reusable_endpoint
logger.warning(
logger.info(
"Reusing existing Model %r (matched model-source tag). "
"No new Model will be created. Pass reuse_resources=False "
"to force a new Model.",
Expand DownExpand Up@@ -5528,7 +5528,7 @@ def deploy(
endpoint_name,
reusable_endpoint,
)
logger.warning(
logger.info(
"Reusing existing endpoint %r (matched model-source tag and "
"deployment configuration). No new resources were created. "
"Pass reuse_resources=False to force a new endpoint.",
Expand Down
50 changes: 36 additions & 14 deletions sagemaker-train/src/sagemaker/train/base_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -399,30 +399,40 @@ def show_metrics(
ValueError: If no training job has been run yet, no logs/metrics
are found, or MLflow is not configured for OSS models.
"""
# Validate that we have a training job to get metrics from
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics."
# Resolve the job reference. Prefer _latest_training_job (CreateTrainingJob),
# fall back to _latest_job (generic CreateJob API used by MTRL).
resolved_job = getattr(self, '_latest_training_job', None)
if resolved_job is None:
latest_job = getattr(self, '_latest_job', None)
if latest_job is None:
raise ValueError(
"No training job found. Call .train() first, then call .show_metrics() "
"to view training metrics. If training has already completed, set the "
"job name directly via trainer._latest_training_job = '<job-name>' or "
"trainer._latest_job = '<job-name>'."
)
resolved_job = (
latest_job.job_name if hasattr(latest_job, 'job_name') else str(latest_job)
)

# Route based on model type
model_name = getattr(self, '_model_name', None)
is_nova = _is_nova_model(model_name) if model_name else False

if is_nova:
return self._show_metrics_cloudwatch(metrics, starting_step, ending_step, start_time, end_time)
return self._show_metrics_cloudwatch(resolved_job, metrics, starting_step, ending_step, start_time, end_time)
else:
return self._show_metrics_mlflow(metrics, starting_step, ending_step)
return self._show_metrics_mlflow(resolved_job, metrics, starting_step, ending_step)

def _show_metrics_mlflow(
self,
resolved_job,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
) -> None:
"""Pull and plot training metrics from MLflow for non-Nova models."""
training_job = self._latest_training_job
training_job = resolved_job

# Resolve the TrainingJob object if it's a string
if isinstance(training_job, str):
Expand DownExpand Up@@ -456,6 +466,7 @@ def _show_metrics_mlflow(

def _show_metrics_cloudwatch(
self,
resolved_job,
metrics: Optional[List[str]] = None,
starting_step: Optional[int] = None,
ending_step: Optional[int] = None,
Expand All@@ -464,7 +475,7 @@ def _show_metrics_cloudwatch(
) -> Any:
"""Parse and plot training metrics from CloudWatch logs (Nova models)."""

training_job = self._latest_training_job
training_job = resolved_job
if hasattr(training_job, 'training_job_name'):
job_id = training_job.training_job_name
elif isinstance(training_job, str):
Expand DownExpand Up@@ -631,10 +642,21 @@ def stream_logs(self, poll: int = 5, start_time: Optional[Any] = None) -> None:
Raises:
ValueError: If no training job has been run yet.
"""
if not hasattr(self, '_latest_training_job') or self._latest_training_job is None:
raise ValueError(
"No training job found. Call .train(wait=False) first, "
"then call .stream_logs() to stream logs in real-time."
# Resolve the job reference. Prefer _latest_training_job (CreateTrainingJob),
# fall back to _latest_job (generic CreateJob API used by MTRL).
resolved_job = getattr(self, '_latest_training_job', None)
if resolved_job is None:
latest_job = getattr(self, '_latest_job', None)
if latest_job is None:
raise ValueError(
"No training job found. Call .train(wait=False) first, "
"then call .stream_logs() to stream logs in real-time. "
"If training has already completed, set the job name directly via "
"trainer._latest_training_job = '<job-name>' or "
"trainer._latest_job = '<job-name>'."
)
resolved_job = (
latest_job.job_name if hasattr(latest_job, 'job_name') else str(latest_job)
)

# Resolve start_time for SMHP jobs
Expand All@@ -645,7 +667,7 @@ def stream_logs(self, poll: int = 5, start_time: Optional[Any] = None) -> None:
else:
start_time_ms = int(start_time)

training_job = self._latest_training_job
training_job = resolved_job
compute = getattr(self, 'compute', None)

if isinstance(compute, HyperPodCompute):
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -35,11 +35,13 @@
"SFT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"CPT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"RLVR": {"reward_score": SMTJ_RLVR_REWARD_SCORE_REGEX},
"MTRL": {"reward_score": SMTJ_RLVR_REWARD_SCORE_REGEX},
},
"smhp": {
"SFT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"CPT": {"training_loss": TRAINING_LOSS_REGEX, "lr": LEARNING_RATE_REGEX},
"RLVR": {"reward_score": SMHP_RLVR_REWARD_SCORE_REGEX},
"MTRL": {"reward_score": SMHP_RLVR_REWARD_SCORE_REGEX},
},
}

Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -248,7 +248,7 @@ def is_multimodal_data(dataset: Union[str, "DataSet"]) -> bool:
True if multimodal fields detected, False otherwise
"""

logger.info(f"Auto-detecting whether dataset is multimodal: {dataset}")
logger.debug(f"Auto-detecting whether dataset is multimodal: {dataset}")

if isinstance(dataset, DataSet):
data_s3_path = dataset.source
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/cpt_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -40,7 +40,6 @@
from sagemaker.core.telemetry.constants import Feature

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class CPTTrainer(BaseTrainer):
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/dpo_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -31,7 +31,6 @@
from sagemaker.train.constants import get_sagemaker_hub_name

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class DPOTrainer(BaseTrainer):
Expand Down
2 changes: 2 additions & 0 deletions sagemaker-train/src/sagemaker/train/multi_turn_rl_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -173,6 +173,8 @@ class MultiTurnRLTrainer(BaseTrainer):
and 'job_name_prefix'. If not specified, no notifications are sent.
"""

_customization_technique = "MTRL"

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

What's the use of this variable?

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

It's for consistency. All the other trainers have it and will be called in BaseTrainer utils


def __init__(
self,
model: Union[str, ModelPackage],
Expand Down
1 change: 0 additions & 1 deletion sagemaker-train/src/sagemaker/train/sft_trainer.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -41,7 +41,6 @@
from sagemaker.core.training.constants import TrainingPlatform

logger = logging.getLogger(__name__)
logger.setLevel(logging.INFO)


class SFTTrainer(BaseTrainer):
Expand Down