loading pre-trained weights for keras model is not supported in distributed training #264

Description

@WuyangLI

System Information

  • Framework (e.g. TensorFlow) / Algorithm (e.g. KMeans): Tensorflow
  • Framework Version: 1.8
  • Python Version: 2
  • CPU or GPU: GPU
  • Python SDK Version: 1.5.1
  • Are you using a custom image: No

Describe the problem

I created a distributed training job which trains a transfer learning model using VGG16.
The job would succeed if I don't load pre-trained weights when creating VGG16 backbone model.

 backend = tf.keras.applications.vgg16.VGG16(weights=None, include_top=False,
input_shape=(224, 224, 3))

However, exception would throw when I try to load pre-trained weights as done in the following code snippet:

 backend = tf.keras.applications.vgg16.VGG16(weights='imagenet', include_top=False,
input_shape=(224, 224, 3))

Note that, for non-distributed training, loading pre-trained weights would not cause any exception.

Minimal repro / logs

InvalidArgumentError (see above for traceback): Cannot assign a device for operation 'block5_conv3/bias': Operation was explicitly assigned to /job:ps/task:0 but available devices are [ /job:localhost/replica:0/task:0/device:CPU:0 ]. Make sure the device specification refers to a valid device.
#011 [[Node: block5_conv3/bias = VariableV2[_class=["loc:@block5_conv3/bias"], container="", dtype=DT_FLOAT, shape=[512], shared_name="", _device="/job:ps/task:0"]()]]
Traceback (most recent call last):
File "/usr/local/lib/python2.7/dist-packages/container_support/training.py", line 36, in start
fw.train()
File "/usr/local/lib/python2.7/dist-packages/tf_container/train_entry_point.py", line 164, in train
train_wrapper.train()
File "/usr/local/lib/python2.7/dist-packages/tf_container/trainer.py", line 73, in train
tf.estimator.train_and_evaluate(estimator=estimator, train_spec=train_spec, eval_spec=eval_spec)
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 439, in train_and_evaluate
executor.run()
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 546, in run
getattr(self, task_to_run)()
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 601, in run_master
self._start_distributed_training(saving_listeners=saving_listeners)
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 739, in _start_distributed_training
saving_listeners=saving_listeners)
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 363, in train
loss = self._train_model(input_fn, hooks, saving_listeners)
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 843, in _train_model
return self._train_model_default(input_fn, hooks, saving_listeners)
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 856, in _train_model_default
features, labels, model_fn_lib.ModeKeys.TRAIN, self.config)
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 831, in _call_model_fn
model_fn_results = self._model_fn(features=features, **kwargs)
File "/usr/local/lib/python2.7/dist-packages/tf_container/trainer.py", line 108, in _model_fn
return self.customer_script.model_fn(features, labels, mode, params)
File "/opt/ml/code/keras_distributed_transfer_learning.py", line 14, in model_fn
input_shape=(224, 224, 3))
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/applications/vgg16.py", line 225, in VGG16
model.load_weights(weights_path)
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/engine/network.py", line 1190, in load_weights
saving.load_weights_from_hdf5_group(f, self.layers)
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/engine/saving.py", line 719, in load_weights_from_hdf5_group
K.batch_set_value(weight_value_tuples)
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 2707, in batch_set_value
get_session().run(assign_ops, feed_dict=feed_dict)
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 442, in get_session
_initialize_variables(session)
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 666, in _initialize_variables
[variables_module.is_variable_initialized(v) for v in candidate_vars])
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 900, in run
run_metadata_ptr)
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1135, in _run
feed_dict_tensor, options, run_metadata)
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1316, in _do_run
run_metadata)
File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1335, in _do_call
raise type(e)(node_def, op, message)
  • Exact command to reproduce:
    code for creating the job
import sagemaker
from sagemaker.tensorflow import TensorFlow
from sagemaker.session import s3_input
from sagemaker import get_execution_role
sagemaker_session = sagemaker.Session()
role = get_execution_role()
training_steps = 100
evaluation_steps = 10
estimator = TensorFlow(
entry_point='keras_distributed_transfer_learning.py',
source_dir='./',
role=role,
training_steps=100,
evaluation_steps=10,
train_instance_count=2,
train_instance_type='ml.p2.xlarge',
input_mode='File')
input_dataset = s3_input('s3://xxxx/cats_and_dogs/')
estimator.fit(input_dataset)

keras_distributed_transfer_learning.py

import tensorflow as tf
from tensorflow.python.estimator.model_fn import ModeKeys as Modes
INPUT_TENSOR_NAME = "input_1"
NUM_CLASSES = 2
BATCH_SIZE = 10
def model_fn(features, labels, mode, params):
"""The model_fn argument for creating an Estimator."""
backend = tf.keras.applications.vgg16.VGG16(weights='imagenet', include_top=False,
input_shape=(224, 224, 3))
x = backend.output
x = tf.keras.layers.Flatten()(x)
x = tf.keras.layers.Dense(NUM_CLASSES, activation='softmax')(x)
model = tf.keras.models.Model(inputs=backend.input, outputs=x)
image = tf.keras.layers.Input(tensor=features[INPUT_TENSOR_NAME])
# Define operations
if mode in (Modes.PREDICT, Modes.EVAL):
logits = model(image, training=False)
predicted_indices = tf.argmax(input=logits, axis=1)
probabilities = tf.nn.softmax(logits, name='softmax_tensor')
if mode in (Modes.TRAIN):
logits = model(image, training=True)
global_step = tf.train.get_or_create_global_step()
loss = tf.losses.softmax_cross_entropy(
onehot_labels=labels, logits=logits)
tf.summary.scalar('OptimizeLoss', loss)
if mode in (Modes.EVAL):
logits = model(image, training=False)
global_step = tf.train.get_or_create_global_step()
loss = tf.losses.softmax_cross_entropy(
onehot_labels=labels, logits=logits)
tf.summary.scalar('OptimizeLoss', loss)
if mode == Modes.PREDICT:
predictions = {
'classes': predicted_indices,
'probabilities': probabilities
}
export_outputs = {
'predictions': tf.estimator.export.PredictOutput(predictions)
}
return tf.estimator.EstimatorSpec(
mode, predictions=predictions, export_outputs=export_outputs)
if mode == Modes.TRAIN:
optimizer = tf.train.AdamOptimizer(learning_rate=0.001)
train_op = optimizer.minimize(loss, global_step=global_step)
return tf.estimator.EstimatorSpec(mode, loss=loss, train_op=train_op)
if mode == Modes.EVAL:
eval_metric_ops = {
'accuracy': tf.metrics.accuracy(tf.argmax(labels, 1), predicted_indices)
}
return tf.estimator.EstimatorSpec(
mode, loss=loss, eval_metric_ops=eval_metric_ops)
def _input_fn(training_dir, input_shape, batch_size):
generator = tf.keras.preprocessing.image.ImageDataGenerator().flow_from_directory(training_dir, target_size=input_shape, batch_size=batch_size)
tensor_shapes = (tf.TensorShape([None, input_shape[0], input_shape[1], 3]), tf.TensorShape([None, NUM_CLASSES]))
tensor_types = (tf.float32, tf.float32)
dataset = tf.data.Dataset.from_generator(lambda: generator, tensor_types, tensor_shapes)
features, labels = dataset.make_one_shot_iterator().get_next()
return {INPUT_TENSOR_NAME: features}, labels
def train_input_fn(training_dir, hyperparameters):
return _input_fn(training_dir + '/train/', (224, 224), BATCH_SIZE)
def eval_input_fn(training_dir, hyperparameters):
return _input_fn(training_dir + '/test/', (224, 224), BATCH_SIZE)
def serving_input_fn(hyperparameters):
inputs = {INPUT_TENSOR_NAME: tf.placeholder(tf.float32, [None, 224, 224, 3])}
return tf.estimator.export.ServingInputReceiver(inputs, inputs)

Activity

Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment

Metadata

Metadata

Assignees

No one assigned

    Labels

    No labels
    No labels

    Type

    No type

    Projects

    No projects

      Milestone

      No milestone

      Relationships

      None yet

      Development

      No branches or pull requests

      Issue actions

      , 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
       blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
      }
      } catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
      })();
      (function(){
      try {
      var __m = "github.com";
      var __re = new RegExp('^' + "github\\.com" + '
      
      Skip to content

      loading pre-trained weights for keras model is not supported in distributed training #264

      Description

      @WuyangLI

      System Information

      • Framework (e.g. TensorFlow) / Algorithm (e.g. KMeans): Tensorflow
      • Framework Version: 1.8
      • Python Version: 2
      • CPU or GPU: GPU
      • Python SDK Version: 1.5.1
      • Are you using a custom image: No

      Describe the problem

      I created a distributed training job which trains a transfer learning model using VGG16.
      The job would succeed if I don't load pre-trained weights when creating VGG16 backbone model.

       backend = tf.keras.applications.vgg16.VGG16(weights=None, include_top=False,
      input_shape=(224, 224, 3))
      

      However, exception would throw when I try to load pre-trained weights as done in the following code snippet:

       backend = tf.keras.applications.vgg16.VGG16(weights='imagenet', include_top=False,
      input_shape=(224, 224, 3))
      

      Note that, for non-distributed training, loading pre-trained weights would not cause any exception.

      Minimal repro / logs

      InvalidArgumentError (see above for traceback): Cannot assign a device for operation 'block5_conv3/bias': Operation was explicitly assigned to /job:ps/task:0 but available devices are [ /job:localhost/replica:0/task:0/device:CPU:0 ]. Make sure the device specification refers to a valid device.
      #011 [[Node: block5_conv3/bias = VariableV2[_class=["loc:@block5_conv3/bias"], container="", dtype=DT_FLOAT, shape=[512], shared_name="", _device="/job:ps/task:0"]()]]
      Traceback (most recent call last):
      File "/usr/local/lib/python2.7/dist-packages/container_support/training.py", line 36, in start
      fw.train()
      File "/usr/local/lib/python2.7/dist-packages/tf_container/train_entry_point.py", line 164, in train
      train_wrapper.train()
      File "/usr/local/lib/python2.7/dist-packages/tf_container/trainer.py", line 73, in train
      tf.estimator.train_and_evaluate(estimator=estimator, train_spec=train_spec, eval_spec=eval_spec)
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 439, in train_and_evaluate
      executor.run()
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 546, in run
      getattr(self, task_to_run)()
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 601, in run_master
      self._start_distributed_training(saving_listeners=saving_listeners)
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 739, in _start_distributed_training
      saving_listeners=saving_listeners)
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 363, in train
      loss = self._train_model(input_fn, hooks, saving_listeners)
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 843, in _train_model
      return self._train_model_default(input_fn, hooks, saving_listeners)
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 856, in _train_model_default
      features, labels, model_fn_lib.ModeKeys.TRAIN, self.config)
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 831, in _call_model_fn
      model_fn_results = self._model_fn(features=features, **kwargs)
      File "/usr/local/lib/python2.7/dist-packages/tf_container/trainer.py", line 108, in _model_fn
      return self.customer_script.model_fn(features, labels, mode, params)
      File "/opt/ml/code/keras_distributed_transfer_learning.py", line 14, in model_fn
      input_shape=(224, 224, 3))
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/applications/vgg16.py", line 225, in VGG16
      model.load_weights(weights_path)
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/engine/network.py", line 1190, in load_weights
      saving.load_weights_from_hdf5_group(f, self.layers)
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/engine/saving.py", line 719, in load_weights_from_hdf5_group
      K.batch_set_value(weight_value_tuples)
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 2707, in batch_set_value
      get_session().run(assign_ops, feed_dict=feed_dict)
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 442, in get_session
      _initialize_variables(session)
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 666, in _initialize_variables
      [variables_module.is_variable_initialized(v) for v in candidate_vars])
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 900, in run
      run_metadata_ptr)
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1135, in _run
      feed_dict_tensor, options, run_metadata)
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1316, in _do_run
      run_metadata)
      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1335, in _do_call
      raise type(e)(node_def, op, message)
      
      • Exact command to reproduce:
        code for creating the job
      import sagemaker
      from sagemaker.tensorflow import TensorFlow
      from sagemaker.session import s3_input
      from sagemaker import get_execution_role
      sagemaker_session = sagemaker.Session()
      role = get_execution_role()
      training_steps = 100
      evaluation_steps = 10
      estimator = TensorFlow(
      entry_point='keras_distributed_transfer_learning.py',
      source_dir='./',
      role=role,
      training_steps=100,
      evaluation_steps=10,
      train_instance_count=2,
      train_instance_type='ml.p2.xlarge',
      input_mode='File')
      input_dataset = s3_input('s3://xxxx/cats_and_dogs/')
      estimator.fit(input_dataset)
      

      keras_distributed_transfer_learning.py

      import tensorflow as tf
      from tensorflow.python.estimator.model_fn import ModeKeys as Modes
      INPUT_TENSOR_NAME = "input_1"
      NUM_CLASSES = 2
      BATCH_SIZE = 10
      def model_fn(features, labels, mode, params):
      """The model_fn argument for creating an Estimator."""
      backend = tf.keras.applications.vgg16.VGG16(weights='imagenet', include_top=False,
      input_shape=(224, 224, 3))
      x = backend.output
      x = tf.keras.layers.Flatten()(x)
      x = tf.keras.layers.Dense(NUM_CLASSES, activation='softmax')(x)
      model = tf.keras.models.Model(inputs=backend.input, outputs=x)
      image = tf.keras.layers.Input(tensor=features[INPUT_TENSOR_NAME])
      # Define operations
      if mode in (Modes.PREDICT, Modes.EVAL):
      logits = model(image, training=False)
      predicted_indices = tf.argmax(input=logits, axis=1)
      probabilities = tf.nn.softmax(logits, name='softmax_tensor')
      if mode in (Modes.TRAIN):
      logits = model(image, training=True)
      global_step = tf.train.get_or_create_global_step()
      loss = tf.losses.softmax_cross_entropy(
      onehot_labels=labels, logits=logits)
      tf.summary.scalar('OptimizeLoss', loss)
      if mode in (Modes.EVAL):
      logits = model(image, training=False)
      global_step = tf.train.get_or_create_global_step()
      loss = tf.losses.softmax_cross_entropy(
      onehot_labels=labels, logits=logits)
      tf.summary.scalar('OptimizeLoss', loss)
      if mode == Modes.PREDICT:
      predictions = {
      'classes': predicted_indices,
      'probabilities': probabilities
      }
      export_outputs = {
      'predictions': tf.estimator.export.PredictOutput(predictions)
      }
      return tf.estimator.EstimatorSpec(
      mode, predictions=predictions, export_outputs=export_outputs)
      if mode == Modes.TRAIN:
      optimizer = tf.train.AdamOptimizer(learning_rate=0.001)
      train_op = optimizer.minimize(loss, global_step=global_step)
      return tf.estimator.EstimatorSpec(mode, loss=loss, train_op=train_op)
      if mode == Modes.EVAL:
      eval_metric_ops = {
      'accuracy': tf.metrics.accuracy(tf.argmax(labels, 1), predicted_indices)
      }
      return tf.estimator.EstimatorSpec(
      mode, loss=loss, eval_metric_ops=eval_metric_ops)
      def _input_fn(training_dir, input_shape, batch_size):
      generator = tf.keras.preprocessing.image.ImageDataGenerator().flow_from_directory(training_dir, target_size=input_shape, batch_size=batch_size)
      tensor_shapes = (tf.TensorShape([None, input_shape[0], input_shape[1], 3]), tf.TensorShape([None, NUM_CLASSES]))
      tensor_types = (tf.float32, tf.float32)
      dataset = tf.data.Dataset.from_generator(lambda: generator, tensor_types, tensor_shapes)
      features, labels = dataset.make_one_shot_iterator().get_next()
      return {INPUT_TENSOR_NAME: features}, labels
      def train_input_fn(training_dir, hyperparameters):
      return _input_fn(training_dir + '/train/', (224, 224), BATCH_SIZE)
      def eval_input_fn(training_dir, hyperparameters):
      return _input_fn(training_dir + '/test/', (224, 224), BATCH_SIZE)
      def serving_input_fn(hyperparameters):
      inputs = {INPUT_TENSOR_NAME: tf.placeholder(tf.float32, [None, 224, 224, 3])}
      return tf.estimator.export.ServingInputReceiver(inputs, inputs)
      

      Activity

      Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment

      Metadata

      Metadata

      Assignees

      No one assigned

        Labels

        No labels
        No labels

        Type

        No type

        Projects

        No projects

          Milestone

          No milestone

          Relationships

          None yet

          Development

          No branches or pull requests

          Issue actions

          , 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
          Skip to content

          loading pre-trained weights for keras model is not supported in distributed training #264

          Description

          @WuyangLI

          System Information

          • Framework (e.g. TensorFlow) / Algorithm (e.g. KMeans): Tensorflow
          • Framework Version: 1.8
          • Python Version: 2
          • CPU or GPU: GPU
          • Python SDK Version: 1.5.1
          • Are you using a custom image: No

          Describe the problem

          I created a distributed training job which trains a transfer learning model using VGG16.
          The job would succeed if I don't load pre-trained weights when creating VGG16 backbone model.

           backend = tf.keras.applications.vgg16.VGG16(weights=None, include_top=False,
          input_shape=(224, 224, 3))
          

          However, exception would throw when I try to load pre-trained weights as done in the following code snippet:

           backend = tf.keras.applications.vgg16.VGG16(weights='imagenet', include_top=False,
          input_shape=(224, 224, 3))
          

          Note that, for non-distributed training, loading pre-trained weights would not cause any exception.

          Minimal repro / logs

          InvalidArgumentError (see above for traceback): Cannot assign a device for operation 'block5_conv3/bias': Operation was explicitly assigned to /job:ps/task:0 but available devices are [ /job:localhost/replica:0/task:0/device:CPU:0 ]. Make sure the device specification refers to a valid device.
          #011 [[Node: block5_conv3/bias = VariableV2[_class=["loc:@block5_conv3/bias"], container="", dtype=DT_FLOAT, shape=[512], shared_name="", _device="/job:ps/task:0"]()]]
          Traceback (most recent call last):
          File "/usr/local/lib/python2.7/dist-packages/container_support/training.py", line 36, in start
          fw.train()
          File "/usr/local/lib/python2.7/dist-packages/tf_container/train_entry_point.py", line 164, in train
          train_wrapper.train()
          File "/usr/local/lib/python2.7/dist-packages/tf_container/trainer.py", line 73, in train
          tf.estimator.train_and_evaluate(estimator=estimator, train_spec=train_spec, eval_spec=eval_spec)
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 439, in train_and_evaluate
          executor.run()
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 546, in run
          getattr(self, task_to_run)()
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 601, in run_master
          self._start_distributed_training(saving_listeners=saving_listeners)
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 739, in _start_distributed_training
          saving_listeners=saving_listeners)
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 363, in train
          loss = self._train_model(input_fn, hooks, saving_listeners)
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 843, in _train_model
          return self._train_model_default(input_fn, hooks, saving_listeners)
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 856, in _train_model_default
          features, labels, model_fn_lib.ModeKeys.TRAIN, self.config)
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 831, in _call_model_fn
          model_fn_results = self._model_fn(features=features, **kwargs)
          File "/usr/local/lib/python2.7/dist-packages/tf_container/trainer.py", line 108, in _model_fn
          return self.customer_script.model_fn(features, labels, mode, params)
          File "/opt/ml/code/keras_distributed_transfer_learning.py", line 14, in model_fn
          input_shape=(224, 224, 3))
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/applications/vgg16.py", line 225, in VGG16
          model.load_weights(weights_path)
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/engine/network.py", line 1190, in load_weights
          saving.load_weights_from_hdf5_group(f, self.layers)
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/engine/saving.py", line 719, in load_weights_from_hdf5_group
          K.batch_set_value(weight_value_tuples)
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 2707, in batch_set_value
          get_session().run(assign_ops, feed_dict=feed_dict)
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 442, in get_session
          _initialize_variables(session)
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 666, in _initialize_variables
          [variables_module.is_variable_initialized(v) for v in candidate_vars])
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 900, in run
          run_metadata_ptr)
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1135, in _run
          feed_dict_tensor, options, run_metadata)
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1316, in _do_run
          run_metadata)
          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1335, in _do_call
          raise type(e)(node_def, op, message)
          
          • Exact command to reproduce:
            code for creating the job
          import sagemaker
          from sagemaker.tensorflow import TensorFlow
          from sagemaker.session import s3_input
          from sagemaker import get_execution_role
          sagemaker_session = sagemaker.Session()
          role = get_execution_role()
          training_steps = 100
          evaluation_steps = 10
          estimator = TensorFlow(
          entry_point='keras_distributed_transfer_learning.py',
          source_dir='./',
          role=role,
          training_steps=100,
          evaluation_steps=10,
          train_instance_count=2,
          train_instance_type='ml.p2.xlarge',
          input_mode='File')
          input_dataset = s3_input('s3://xxxx/cats_and_dogs/')
          estimator.fit(input_dataset)
          

          keras_distributed_transfer_learning.py

          import tensorflow as tf
          from tensorflow.python.estimator.model_fn import ModeKeys as Modes
          INPUT_TENSOR_NAME = "input_1"
          NUM_CLASSES = 2
          BATCH_SIZE = 10
          def model_fn(features, labels, mode, params):
          """The model_fn argument for creating an Estimator."""
          backend = tf.keras.applications.vgg16.VGG16(weights='imagenet', include_top=False,
          input_shape=(224, 224, 3))
          x = backend.output
          x = tf.keras.layers.Flatten()(x)
          x = tf.keras.layers.Dense(NUM_CLASSES, activation='softmax')(x)
          model = tf.keras.models.Model(inputs=backend.input, outputs=x)
          image = tf.keras.layers.Input(tensor=features[INPUT_TENSOR_NAME])
          # Define operations
          if mode in (Modes.PREDICT, Modes.EVAL):
          logits = model(image, training=False)
          predicted_indices = tf.argmax(input=logits, axis=1)
          probabilities = tf.nn.softmax(logits, name='softmax_tensor')
          if mode in (Modes.TRAIN):
          logits = model(image, training=True)
          global_step = tf.train.get_or_create_global_step()
          loss = tf.losses.softmax_cross_entropy(
          onehot_labels=labels, logits=logits)
          tf.summary.scalar('OptimizeLoss', loss)
          if mode in (Modes.EVAL):
          logits = model(image, training=False)
          global_step = tf.train.get_or_create_global_step()
          loss = tf.losses.softmax_cross_entropy(
          onehot_labels=labels, logits=logits)
          tf.summary.scalar('OptimizeLoss', loss)
          if mode == Modes.PREDICT:
          predictions = {
          'classes': predicted_indices,
          'probabilities': probabilities
          }
          export_outputs = {
          'predictions': tf.estimator.export.PredictOutput(predictions)
          }
          return tf.estimator.EstimatorSpec(
          mode, predictions=predictions, export_outputs=export_outputs)
          if mode == Modes.TRAIN:
          optimizer = tf.train.AdamOptimizer(learning_rate=0.001)
          train_op = optimizer.minimize(loss, global_step=global_step)
          return tf.estimator.EstimatorSpec(mode, loss=loss, train_op=train_op)
          if mode == Modes.EVAL:
          eval_metric_ops = {
          'accuracy': tf.metrics.accuracy(tf.argmax(labels, 1), predicted_indices)
          }
          return tf.estimator.EstimatorSpec(
          mode, loss=loss, eval_metric_ops=eval_metric_ops)
          def _input_fn(training_dir, input_shape, batch_size):
          generator = tf.keras.preprocessing.image.ImageDataGenerator().flow_from_directory(training_dir, target_size=input_shape, batch_size=batch_size)
          tensor_shapes = (tf.TensorShape([None, input_shape[0], input_shape[1], 3]), tf.TensorShape([None, NUM_CLASSES]))
          tensor_types = (tf.float32, tf.float32)
          dataset = tf.data.Dataset.from_generator(lambda: generator, tensor_types, tensor_shapes)
          features, labels = dataset.make_one_shot_iterator().get_next()
          return {INPUT_TENSOR_NAME: features}, labels
          def train_input_fn(training_dir, hyperparameters):
          return _input_fn(training_dir + '/train/', (224, 224), BATCH_SIZE)
          def eval_input_fn(training_dir, hyperparameters):
          return _input_fn(training_dir + '/test/', (224, 224), BATCH_SIZE)
          def serving_input_fn(hyperparameters):
          inputs = {INPUT_TENSOR_NAME: tf.placeholder(tf.float32, [None, 224, 224, 3])}
          return tf.estimator.export.ServingInputReceiver(inputs, inputs)
          

          Activity

          Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment

          Metadata

          Metadata

          Assignees

          No one assigned

            Labels

            No labels
            No labels

            Type

            No type

            Projects

            No projects

              Milestone

              No milestone

              Relationships

              None yet

              Development

              No branches or pull requests

              Issue actions

              , 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
              Skip to content

              loading pre-trained weights for keras model is not supported in distributed training #264

              Description

              @WuyangLI

              System Information

              • Framework (e.g. TensorFlow) / Algorithm (e.g. KMeans): Tensorflow
              • Framework Version: 1.8
              • Python Version: 2
              • CPU or GPU: GPU
              • Python SDK Version: 1.5.1
              • Are you using a custom image: No

              Describe the problem

              I created a distributed training job which trains a transfer learning model using VGG16.
              The job would succeed if I don't load pre-trained weights when creating VGG16 backbone model.

               backend = tf.keras.applications.vgg16.VGG16(weights=None, include_top=False,
              input_shape=(224, 224, 3))
              

              However, exception would throw when I try to load pre-trained weights as done in the following code snippet:

               backend = tf.keras.applications.vgg16.VGG16(weights='imagenet', include_top=False,
              input_shape=(224, 224, 3))
              

              Note that, for non-distributed training, loading pre-trained weights would not cause any exception.

              Minimal repro / logs

              InvalidArgumentError (see above for traceback): Cannot assign a device for operation 'block5_conv3/bias': Operation was explicitly assigned to /job:ps/task:0 but available devices are [ /job:localhost/replica:0/task:0/device:CPU:0 ]. Make sure the device specification refers to a valid device.
              #011 [[Node: block5_conv3/bias = VariableV2[_class=["loc:@block5_conv3/bias"], container="", dtype=DT_FLOAT, shape=[512], shared_name="", _device="/job:ps/task:0"]()]]
              Traceback (most recent call last):
              File "/usr/local/lib/python2.7/dist-packages/container_support/training.py", line 36, in start
              fw.train()
              File "/usr/local/lib/python2.7/dist-packages/tf_container/train_entry_point.py", line 164, in train
              train_wrapper.train()
              File "/usr/local/lib/python2.7/dist-packages/tf_container/trainer.py", line 73, in train
              tf.estimator.train_and_evaluate(estimator=estimator, train_spec=train_spec, eval_spec=eval_spec)
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 439, in train_and_evaluate
              executor.run()
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 546, in run
              getattr(self, task_to_run)()
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 601, in run_master
              self._start_distributed_training(saving_listeners=saving_listeners)
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 739, in _start_distributed_training
              saving_listeners=saving_listeners)
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 363, in train
              loss = self._train_model(input_fn, hooks, saving_listeners)
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 843, in _train_model
              return self._train_model_default(input_fn, hooks, saving_listeners)
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 856, in _train_model_default
              features, labels, model_fn_lib.ModeKeys.TRAIN, self.config)
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 831, in _call_model_fn
              model_fn_results = self._model_fn(features=features, **kwargs)
              File "/usr/local/lib/python2.7/dist-packages/tf_container/trainer.py", line 108, in _model_fn
              return self.customer_script.model_fn(features, labels, mode, params)
              File "/opt/ml/code/keras_distributed_transfer_learning.py", line 14, in model_fn
              input_shape=(224, 224, 3))
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/applications/vgg16.py", line 225, in VGG16
              model.load_weights(weights_path)
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/engine/network.py", line 1190, in load_weights
              saving.load_weights_from_hdf5_group(f, self.layers)
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/engine/saving.py", line 719, in load_weights_from_hdf5_group
              K.batch_set_value(weight_value_tuples)
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 2707, in batch_set_value
              get_session().run(assign_ops, feed_dict=feed_dict)
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 442, in get_session
              _initialize_variables(session)
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 666, in _initialize_variables
              [variables_module.is_variable_initialized(v) for v in candidate_vars])
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 900, in run
              run_metadata_ptr)
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1135, in _run
              feed_dict_tensor, options, run_metadata)
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1316, in _do_run
              run_metadata)
              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1335, in _do_call
              raise type(e)(node_def, op, message)
              
              • Exact command to reproduce:
                code for creating the job
              import sagemaker
              from sagemaker.tensorflow import TensorFlow
              from sagemaker.session import s3_input
              from sagemaker import get_execution_role
              sagemaker_session = sagemaker.Session()
              role = get_execution_role()
              training_steps = 100
              evaluation_steps = 10
              estimator = TensorFlow(
              entry_point='keras_distributed_transfer_learning.py',
              source_dir='./',
              role=role,
              training_steps=100,
              evaluation_steps=10,
              train_instance_count=2,
              train_instance_type='ml.p2.xlarge',
              input_mode='File')
              input_dataset = s3_input('s3://xxxx/cats_and_dogs/')
              estimator.fit(input_dataset)
              

              keras_distributed_transfer_learning.py

              import tensorflow as tf
              from tensorflow.python.estimator.model_fn import ModeKeys as Modes
              INPUT_TENSOR_NAME = "input_1"
              NUM_CLASSES = 2
              BATCH_SIZE = 10
              def model_fn(features, labels, mode, params):
              """The model_fn argument for creating an Estimator."""
              backend = tf.keras.applications.vgg16.VGG16(weights='imagenet', include_top=False,
              input_shape=(224, 224, 3))
              x = backend.output
              x = tf.keras.layers.Flatten()(x)
              x = tf.keras.layers.Dense(NUM_CLASSES, activation='softmax')(x)
              model = tf.keras.models.Model(inputs=backend.input, outputs=x)
              image = tf.keras.layers.Input(tensor=features[INPUT_TENSOR_NAME])
              # Define operations
              if mode in (Modes.PREDICT, Modes.EVAL):
              logits = model(image, training=False)
              predicted_indices = tf.argmax(input=logits, axis=1)
              probabilities = tf.nn.softmax(logits, name='softmax_tensor')
              if mode in (Modes.TRAIN):
              logits = model(image, training=True)
              global_step = tf.train.get_or_create_global_step()
              loss = tf.losses.softmax_cross_entropy(
              onehot_labels=labels, logits=logits)
              tf.summary.scalar('OptimizeLoss', loss)
              if mode in (Modes.EVAL):
              logits = model(image, training=False)
              global_step = tf.train.get_or_create_global_step()
              loss = tf.losses.softmax_cross_entropy(
              onehot_labels=labels, logits=logits)
              tf.summary.scalar('OptimizeLoss', loss)
              if mode == Modes.PREDICT:
              predictions = {
              'classes': predicted_indices,
              'probabilities': probabilities
              }
              export_outputs = {
              'predictions': tf.estimator.export.PredictOutput(predictions)
              }
              return tf.estimator.EstimatorSpec(
              mode, predictions=predictions, export_outputs=export_outputs)
              if mode == Modes.TRAIN:
              optimizer = tf.train.AdamOptimizer(learning_rate=0.001)
              train_op = optimizer.minimize(loss, global_step=global_step)
              return tf.estimator.EstimatorSpec(mode, loss=loss, train_op=train_op)
              if mode == Modes.EVAL:
              eval_metric_ops = {
              'accuracy': tf.metrics.accuracy(tf.argmax(labels, 1), predicted_indices)
              }
              return tf.estimator.EstimatorSpec(
              mode, loss=loss, eval_metric_ops=eval_metric_ops)
              def _input_fn(training_dir, input_shape, batch_size):
              generator = tf.keras.preprocessing.image.ImageDataGenerator().flow_from_directory(training_dir, target_size=input_shape, batch_size=batch_size)
              tensor_shapes = (tf.TensorShape([None, input_shape[0], input_shape[1], 3]), tf.TensorShape([None, NUM_CLASSES]))
              tensor_types = (tf.float32, tf.float32)
              dataset = tf.data.Dataset.from_generator(lambda: generator, tensor_types, tensor_shapes)
              features, labels = dataset.make_one_shot_iterator().get_next()
              return {INPUT_TENSOR_NAME: features}, labels
              def train_input_fn(training_dir, hyperparameters):
              return _input_fn(training_dir + '/train/', (224, 224), BATCH_SIZE)
              def eval_input_fn(training_dir, hyperparameters):
              return _input_fn(training_dir + '/test/', (224, 224), BATCH_SIZE)
              def serving_input_fn(hyperparameters):
              inputs = {INPUT_TENSOR_NAME: tf.placeholder(tf.float32, [None, 224, 224, 3])}
              return tf.estimator.export.ServingInputReceiver(inputs, inputs)
              

              Activity

              Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment

              Metadata

              Metadata

              Assignees

              No one assigned

                Labels

                No labels
                No labels

                Type

                No type

                Projects

                No projects

                  Milestone

                  No milestone

                  Relationships

                  None yet

                  Development

                  No branches or pull requests

                  Issue actions

                  , 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
                  Skip to content

                  loading pre-trained weights for keras model is not supported in distributed training #264

                  Description

                  @WuyangLI

                  System Information

                  • Framework (e.g. TensorFlow) / Algorithm (e.g. KMeans): Tensorflow
                  • Framework Version: 1.8
                  • Python Version: 2
                  • CPU or GPU: GPU
                  • Python SDK Version: 1.5.1
                  • Are you using a custom image: No

                  Describe the problem

                  I created a distributed training job which trains a transfer learning model using VGG16.
                  The job would succeed if I don't load pre-trained weights when creating VGG16 backbone model.

                   backend = tf.keras.applications.vgg16.VGG16(weights=None, include_top=False,
                  input_shape=(224, 224, 3))
                  

                  However, exception would throw when I try to load pre-trained weights as done in the following code snippet:

                   backend = tf.keras.applications.vgg16.VGG16(weights='imagenet', include_top=False,
                  input_shape=(224, 224, 3))
                  

                  Note that, for non-distributed training, loading pre-trained weights would not cause any exception.

                  Minimal repro / logs

                  InvalidArgumentError (see above for traceback): Cannot assign a device for operation 'block5_conv3/bias': Operation was explicitly assigned to /job:ps/task:0 but available devices are [ /job:localhost/replica:0/task:0/device:CPU:0 ]. Make sure the device specification refers to a valid device.
                  #011 [[Node: block5_conv3/bias = VariableV2[_class=["loc:@block5_conv3/bias"], container="", dtype=DT_FLOAT, shape=[512], shared_name="", _device="/job:ps/task:0"]()]]
                  Traceback (most recent call last):
                  File "/usr/local/lib/python2.7/dist-packages/container_support/training.py", line 36, in start
                  fw.train()
                  File "/usr/local/lib/python2.7/dist-packages/tf_container/train_entry_point.py", line 164, in train
                  train_wrapper.train()
                  File "/usr/local/lib/python2.7/dist-packages/tf_container/trainer.py", line 73, in train
                  tf.estimator.train_and_evaluate(estimator=estimator, train_spec=train_spec, eval_spec=eval_spec)
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 439, in train_and_evaluate
                  executor.run()
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 546, in run
                  getattr(self, task_to_run)()
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 601, in run_master
                  self._start_distributed_training(saving_listeners=saving_listeners)
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 739, in _start_distributed_training
                  saving_listeners=saving_listeners)
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 363, in train
                  loss = self._train_model(input_fn, hooks, saving_listeners)
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 843, in _train_model
                  return self._train_model_default(input_fn, hooks, saving_listeners)
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 856, in _train_model_default
                  features, labels, model_fn_lib.ModeKeys.TRAIN, self.config)
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 831, in _call_model_fn
                  model_fn_results = self._model_fn(features=features, **kwargs)
                  File "/usr/local/lib/python2.7/dist-packages/tf_container/trainer.py", line 108, in _model_fn
                  return self.customer_script.model_fn(features, labels, mode, params)
                  File "/opt/ml/code/keras_distributed_transfer_learning.py", line 14, in model_fn
                  input_shape=(224, 224, 3))
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/applications/vgg16.py", line 225, in VGG16
                  model.load_weights(weights_path)
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/engine/network.py", line 1190, in load_weights
                  saving.load_weights_from_hdf5_group(f, self.layers)
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/engine/saving.py", line 719, in load_weights_from_hdf5_group
                  K.batch_set_value(weight_value_tuples)
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 2707, in batch_set_value
                  get_session().run(assign_ops, feed_dict=feed_dict)
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 442, in get_session
                  _initialize_variables(session)
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 666, in _initialize_variables
                  [variables_module.is_variable_initialized(v) for v in candidate_vars])
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 900, in run
                  run_metadata_ptr)
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1135, in _run
                  feed_dict_tensor, options, run_metadata)
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1316, in _do_run
                  run_metadata)
                  File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1335, in _do_call
                  raise type(e)(node_def, op, message)
                  
                  • Exact command to reproduce:
                    code for creating the job
                  import sagemaker
                  from sagemaker.tensorflow import TensorFlow
                  from sagemaker.session import s3_input
                  from sagemaker import get_execution_role
                  sagemaker_session = sagemaker.Session()
                  role = get_execution_role()
                  training_steps = 100
                  evaluation_steps = 10
                  estimator = TensorFlow(
                  entry_point='keras_distributed_transfer_learning.py',
                  source_dir='./',
                  role=role,
                  training_steps=100,
                  evaluation_steps=10,
                  train_instance_count=2,
                  train_instance_type='ml.p2.xlarge',
                  input_mode='File')
                  input_dataset = s3_input('s3://xxxx/cats_and_dogs/')
                  estimator.fit(input_dataset)
                  

                  keras_distributed_transfer_learning.py

                  import tensorflow as tf
                  from tensorflow.python.estimator.model_fn import ModeKeys as Modes
                  INPUT_TENSOR_NAME = "input_1"
                  NUM_CLASSES = 2
                  BATCH_SIZE = 10
                  def model_fn(features, labels, mode, params):
                  """The model_fn argument for creating an Estimator."""
                  backend = tf.keras.applications.vgg16.VGG16(weights='imagenet', include_top=False,
                  input_shape=(224, 224, 3))
                  x = backend.output
                  x = tf.keras.layers.Flatten()(x)
                  x = tf.keras.layers.Dense(NUM_CLASSES, activation='softmax')(x)
                  model = tf.keras.models.Model(inputs=backend.input, outputs=x)
                  image = tf.keras.layers.Input(tensor=features[INPUT_TENSOR_NAME])
                  # Define operations
                  if mode in (Modes.PREDICT, Modes.EVAL):
                  logits = model(image, training=False)
                  predicted_indices = tf.argmax(input=logits, axis=1)
                  probabilities = tf.nn.softmax(logits, name='softmax_tensor')
                  if mode in (Modes.TRAIN):
                  logits = model(image, training=True)
                  global_step = tf.train.get_or_create_global_step()
                  loss = tf.losses.softmax_cross_entropy(
                  onehot_labels=labels, logits=logits)
                  tf.summary.scalar('OptimizeLoss', loss)
                  if mode in (Modes.EVAL):
                  logits = model(image, training=False)
                  global_step = tf.train.get_or_create_global_step()
                  loss = tf.losses.softmax_cross_entropy(
                  onehot_labels=labels, logits=logits)
                  tf.summary.scalar('OptimizeLoss', loss)
                  if mode == Modes.PREDICT:
                  predictions = {
                  'classes': predicted_indices,
                  'probabilities': probabilities
                  }
                  export_outputs = {
                  'predictions': tf.estimator.export.PredictOutput(predictions)
                  }
                  return tf.estimator.EstimatorSpec(
                  mode, predictions=predictions, export_outputs=export_outputs)
                  if mode == Modes.TRAIN:
                  optimizer = tf.train.AdamOptimizer(learning_rate=0.001)
                  train_op = optimizer.minimize(loss, global_step=global_step)
                  return tf.estimator.EstimatorSpec(mode, loss=loss, train_op=train_op)
                  if mode == Modes.EVAL:
                  eval_metric_ops = {
                  'accuracy': tf.metrics.accuracy(tf.argmax(labels, 1), predicted_indices)
                  }
                  return tf.estimator.EstimatorSpec(
                  mode, loss=loss, eval_metric_ops=eval_metric_ops)
                  def _input_fn(training_dir, input_shape, batch_size):
                  generator = tf.keras.preprocessing.image.ImageDataGenerator().flow_from_directory(training_dir, target_size=input_shape, batch_size=batch_size)
                  tensor_shapes = (tf.TensorShape([None, input_shape[0], input_shape[1], 3]), tf.TensorShape([None, NUM_CLASSES]))
                  tensor_types = (tf.float32, tf.float32)
                  dataset = tf.data.Dataset.from_generator(lambda: generator, tensor_types, tensor_shapes)
                  features, labels = dataset.make_one_shot_iterator().get_next()
                  return {INPUT_TENSOR_NAME: features}, labels
                  def train_input_fn(training_dir, hyperparameters):
                  return _input_fn(training_dir + '/train/', (224, 224), BATCH_SIZE)
                  def eval_input_fn(training_dir, hyperparameters):
                  return _input_fn(training_dir + '/test/', (224, 224), BATCH_SIZE)
                  def serving_input_fn(hyperparameters):
                  inputs = {INPUT_TENSOR_NAME: tf.placeholder(tf.float32, [None, 224, 224, 3])}
                  return tf.estimator.export.ServingInputReceiver(inputs, inputs)
                  

                  Activity

                  Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment

                  Metadata

                  Metadata

                  Assignees

                  No one assigned

                    Labels

                    No labels
                    No labels

                    Type

                    No type

                    Projects

                    No projects

                      Milestone

                      No milestone

                      Relationships

                      None yet

                      Development

                      No branches or pull requests

                      Issue actions

                      , 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
                      Skip to content

                      loading pre-trained weights for keras model is not supported in distributed training #264

                      Description

                      @WuyangLI

                      System Information

                      • Framework (e.g. TensorFlow) / Algorithm (e.g. KMeans): Tensorflow
                      • Framework Version: 1.8
                      • Python Version: 2
                      • CPU or GPU: GPU
                      • Python SDK Version: 1.5.1
                      • Are you using a custom image: No

                      Describe the problem

                      I created a distributed training job which trains a transfer learning model using VGG16.
                      The job would succeed if I don't load pre-trained weights when creating VGG16 backbone model.

                       backend = tf.keras.applications.vgg16.VGG16(weights=None, include_top=False,
                      input_shape=(224, 224, 3))
                      

                      However, exception would throw when I try to load pre-trained weights as done in the following code snippet:

                       backend = tf.keras.applications.vgg16.VGG16(weights='imagenet', include_top=False,
                      input_shape=(224, 224, 3))
                      

                      Note that, for non-distributed training, loading pre-trained weights would not cause any exception.

                      Minimal repro / logs

                      InvalidArgumentError (see above for traceback): Cannot assign a device for operation 'block5_conv3/bias': Operation was explicitly assigned to /job:ps/task:0 but available devices are [ /job:localhost/replica:0/task:0/device:CPU:0 ]. Make sure the device specification refers to a valid device.
                      #011 [[Node: block5_conv3/bias = VariableV2[_class=["loc:@block5_conv3/bias"], container="", dtype=DT_FLOAT, shape=[512], shared_name="", _device="/job:ps/task:0"]()]]
                      Traceback (most recent call last):
                      File "/usr/local/lib/python2.7/dist-packages/container_support/training.py", line 36, in start
                      fw.train()
                      File "/usr/local/lib/python2.7/dist-packages/tf_container/train_entry_point.py", line 164, in train
                      train_wrapper.train()
                      File "/usr/local/lib/python2.7/dist-packages/tf_container/trainer.py", line 73, in train
                      tf.estimator.train_and_evaluate(estimator=estimator, train_spec=train_spec, eval_spec=eval_spec)
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 439, in train_and_evaluate
                      executor.run()
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 546, in run
                      getattr(self, task_to_run)()
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 601, in run_master
                      self._start_distributed_training(saving_listeners=saving_listeners)
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 739, in _start_distributed_training
                      saving_listeners=saving_listeners)
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 363, in train
                      loss = self._train_model(input_fn, hooks, saving_listeners)
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 843, in _train_model
                      return self._train_model_default(input_fn, hooks, saving_listeners)
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 856, in _train_model_default
                      features, labels, model_fn_lib.ModeKeys.TRAIN, self.config)
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 831, in _call_model_fn
                      model_fn_results = self._model_fn(features=features, **kwargs)
                      File "/usr/local/lib/python2.7/dist-packages/tf_container/trainer.py", line 108, in _model_fn
                      return self.customer_script.model_fn(features, labels, mode, params)
                      File "/opt/ml/code/keras_distributed_transfer_learning.py", line 14, in model_fn
                      input_shape=(224, 224, 3))
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/applications/vgg16.py", line 225, in VGG16
                      model.load_weights(weights_path)
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/engine/network.py", line 1190, in load_weights
                      saving.load_weights_from_hdf5_group(f, self.layers)
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/engine/saving.py", line 719, in load_weights_from_hdf5_group
                      K.batch_set_value(weight_value_tuples)
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 2707, in batch_set_value
                      get_session().run(assign_ops, feed_dict=feed_dict)
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 442, in get_session
                      _initialize_variables(session)
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 666, in _initialize_variables
                      [variables_module.is_variable_initialized(v) for v in candidate_vars])
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 900, in run
                      run_metadata_ptr)
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1135, in _run
                      feed_dict_tensor, options, run_metadata)
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1316, in _do_run
                      run_metadata)
                      File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1335, in _do_call
                      raise type(e)(node_def, op, message)
                      
                      • Exact command to reproduce:
                        code for creating the job
                      import sagemaker
                      from sagemaker.tensorflow import TensorFlow
                      from sagemaker.session import s3_input
                      from sagemaker import get_execution_role
                      sagemaker_session = sagemaker.Session()
                      role = get_execution_role()
                      training_steps = 100
                      evaluation_steps = 10
                      estimator = TensorFlow(
                      entry_point='keras_distributed_transfer_learning.py',
                      source_dir='./',
                      role=role,
                      training_steps=100,
                      evaluation_steps=10,
                      train_instance_count=2,
                      train_instance_type='ml.p2.xlarge',
                      input_mode='File')
                      input_dataset = s3_input('s3://xxxx/cats_and_dogs/')
                      estimator.fit(input_dataset)
                      

                      keras_distributed_transfer_learning.py

                      import tensorflow as tf
                      from tensorflow.python.estimator.model_fn import ModeKeys as Modes
                      INPUT_TENSOR_NAME = "input_1"
                      NUM_CLASSES = 2
                      BATCH_SIZE = 10
                      def model_fn(features, labels, mode, params):
                      """The model_fn argument for creating an Estimator."""
                      backend = tf.keras.applications.vgg16.VGG16(weights='imagenet', include_top=False,
                      input_shape=(224, 224, 3))
                      x = backend.output
                      x = tf.keras.layers.Flatten()(x)
                      x = tf.keras.layers.Dense(NUM_CLASSES, activation='softmax')(x)
                      model = tf.keras.models.Model(inputs=backend.input, outputs=x)
                      image = tf.keras.layers.Input(tensor=features[INPUT_TENSOR_NAME])
                      # Define operations
                      if mode in (Modes.PREDICT, Modes.EVAL):
                      logits = model(image, training=False)
                      predicted_indices = tf.argmax(input=logits, axis=1)
                      probabilities = tf.nn.softmax(logits, name='softmax_tensor')
                      if mode in (Modes.TRAIN):
                      logits = model(image, training=True)
                      global_step = tf.train.get_or_create_global_step()
                      loss = tf.losses.softmax_cross_entropy(
                      onehot_labels=labels, logits=logits)
                      tf.summary.scalar('OptimizeLoss', loss)
                      if mode in (Modes.EVAL):
                      logits = model(image, training=False)
                      global_step = tf.train.get_or_create_global_step()
                      loss = tf.losses.softmax_cross_entropy(
                      onehot_labels=labels, logits=logits)
                      tf.summary.scalar('OptimizeLoss', loss)
                      if mode == Modes.PREDICT:
                      predictions = {
                      'classes': predicted_indices,
                      'probabilities': probabilities
                      }
                      export_outputs = {
                      'predictions': tf.estimator.export.PredictOutput(predictions)
                      }
                      return tf.estimator.EstimatorSpec(
                      mode, predictions=predictions, export_outputs=export_outputs)
                      if mode == Modes.TRAIN:
                      optimizer = tf.train.AdamOptimizer(learning_rate=0.001)
                      train_op = optimizer.minimize(loss, global_step=global_step)
                      return tf.estimator.EstimatorSpec(mode, loss=loss, train_op=train_op)
                      if mode == Modes.EVAL:
                      eval_metric_ops = {
                      'accuracy': tf.metrics.accuracy(tf.argmax(labels, 1), predicted_indices)
                      }
                      return tf.estimator.EstimatorSpec(
                      mode, loss=loss, eval_metric_ops=eval_metric_ops)
                      def _input_fn(training_dir, input_shape, batch_size):
                      generator = tf.keras.preprocessing.image.ImageDataGenerator().flow_from_directory(training_dir, target_size=input_shape, batch_size=batch_size)
                      tensor_shapes = (tf.TensorShape([None, input_shape[0], input_shape[1], 3]), tf.TensorShape([None, NUM_CLASSES]))
                      tensor_types = (tf.float32, tf.float32)
                      dataset = tf.data.Dataset.from_generator(lambda: generator, tensor_types, tensor_shapes)
                      features, labels = dataset.make_one_shot_iterator().get_next()
                      return {INPUT_TENSOR_NAME: features}, labels
                      def train_input_fn(training_dir, hyperparameters):
                      return _input_fn(training_dir + '/train/', (224, 224), BATCH_SIZE)
                      def eval_input_fn(training_dir, hyperparameters):
                      return _input_fn(training_dir + '/test/', (224, 224), BATCH_SIZE)
                      def serving_input_fn(hyperparameters):
                      inputs = {INPUT_TENSOR_NAME: tf.placeholder(tf.float32, [None, 224, 224, 3])}
                      return tf.estimator.export.ServingInputReceiver(inputs, inputs)
                      

                      Activity

                      Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment

                      Metadata

                      Metadata

                      Assignees

                      No one assigned

                        Labels

                        No labels
                        No labels

                        Type

                        No type

                        Projects

                        No projects

                          Milestone

                          No milestone

                          Relationships

                          None yet

                          Development

                          No branches or pull requests

                          Issue actions

                          , 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
                          Skip to content

                          loading pre-trained weights for keras model is not supported in distributed training #264

                          Description

                          @WuyangLI

                          System Information

                          • Framework (e.g. TensorFlow) / Algorithm (e.g. KMeans): Tensorflow
                          • Framework Version: 1.8
                          • Python Version: 2
                          • CPU or GPU: GPU
                          • Python SDK Version: 1.5.1
                          • Are you using a custom image: No

                          Describe the problem

                          I created a distributed training job which trains a transfer learning model using VGG16.
                          The job would succeed if I don't load pre-trained weights when creating VGG16 backbone model.

                           backend = tf.keras.applications.vgg16.VGG16(weights=None, include_top=False,
                          input_shape=(224, 224, 3))
                          

                          However, exception would throw when I try to load pre-trained weights as done in the following code snippet:

                           backend = tf.keras.applications.vgg16.VGG16(weights='imagenet', include_top=False,
                          input_shape=(224, 224, 3))
                          

                          Note that, for non-distributed training, loading pre-trained weights would not cause any exception.

                          Minimal repro / logs

                          InvalidArgumentError (see above for traceback): Cannot assign a device for operation 'block5_conv3/bias': Operation was explicitly assigned to /job:ps/task:0 but available devices are [ /job:localhost/replica:0/task:0/device:CPU:0 ]. Make sure the device specification refers to a valid device.
                          #011 [[Node: block5_conv3/bias = VariableV2[_class=["loc:@block5_conv3/bias"], container="", dtype=DT_FLOAT, shape=[512], shared_name="", _device="/job:ps/task:0"]()]]
                          Traceback (most recent call last):
                          File "/usr/local/lib/python2.7/dist-packages/container_support/training.py", line 36, in start
                          fw.train()
                          File "/usr/local/lib/python2.7/dist-packages/tf_container/train_entry_point.py", line 164, in train
                          train_wrapper.train()
                          File "/usr/local/lib/python2.7/dist-packages/tf_container/trainer.py", line 73, in train
                          tf.estimator.train_and_evaluate(estimator=estimator, train_spec=train_spec, eval_spec=eval_spec)
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 439, in train_and_evaluate
                          executor.run()
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 546, in run
                          getattr(self, task_to_run)()
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 601, in run_master
                          self._start_distributed_training(saving_listeners=saving_listeners)
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 739, in _start_distributed_training
                          saving_listeners=saving_listeners)
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 363, in train
                          loss = self._train_model(input_fn, hooks, saving_listeners)
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 843, in _train_model
                          return self._train_model_default(input_fn, hooks, saving_listeners)
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 856, in _train_model_default
                          features, labels, model_fn_lib.ModeKeys.TRAIN, self.config)
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 831, in _call_model_fn
                          model_fn_results = self._model_fn(features=features, **kwargs)
                          File "/usr/local/lib/python2.7/dist-packages/tf_container/trainer.py", line 108, in _model_fn
                          return self.customer_script.model_fn(features, labels, mode, params)
                          File "/opt/ml/code/keras_distributed_transfer_learning.py", line 14, in model_fn
                          input_shape=(224, 224, 3))
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/applications/vgg16.py", line 225, in VGG16
                          model.load_weights(weights_path)
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/engine/network.py", line 1190, in load_weights
                          saving.load_weights_from_hdf5_group(f, self.layers)
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/engine/saving.py", line 719, in load_weights_from_hdf5_group
                          K.batch_set_value(weight_value_tuples)
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 2707, in batch_set_value
                          get_session().run(assign_ops, feed_dict=feed_dict)
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 442, in get_session
                          _initialize_variables(session)
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 666, in _initialize_variables
                          [variables_module.is_variable_initialized(v) for v in candidate_vars])
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 900, in run
                          run_metadata_ptr)
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1135, in _run
                          feed_dict_tensor, options, run_metadata)
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1316, in _do_run
                          run_metadata)
                          File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1335, in _do_call
                          raise type(e)(node_def, op, message)
                          
                          • Exact command to reproduce:
                            code for creating the job
                          import sagemaker
                          from sagemaker.tensorflow import TensorFlow
                          from sagemaker.session import s3_input
                          from sagemaker import get_execution_role
                          sagemaker_session = sagemaker.Session()
                          role = get_execution_role()
                          training_steps = 100
                          evaluation_steps = 10
                          estimator = TensorFlow(
                          entry_point='keras_distributed_transfer_learning.py',
                          source_dir='./',
                          role=role,
                          training_steps=100,
                          evaluation_steps=10,
                          train_instance_count=2,
                          train_instance_type='ml.p2.xlarge',
                          input_mode='File')
                          input_dataset = s3_input('s3://xxxx/cats_and_dogs/')
                          estimator.fit(input_dataset)
                          

                          keras_distributed_transfer_learning.py

                          import tensorflow as tf
                          from tensorflow.python.estimator.model_fn import ModeKeys as Modes
                          INPUT_TENSOR_NAME = "input_1"
                          NUM_CLASSES = 2
                          BATCH_SIZE = 10
                          def model_fn(features, labels, mode, params):
                          """The model_fn argument for creating an Estimator."""
                          backend = tf.keras.applications.vgg16.VGG16(weights='imagenet', include_top=False,
                          input_shape=(224, 224, 3))
                          x = backend.output
                          x = tf.keras.layers.Flatten()(x)
                          x = tf.keras.layers.Dense(NUM_CLASSES, activation='softmax')(x)
                          model = tf.keras.models.Model(inputs=backend.input, outputs=x)
                          image = tf.keras.layers.Input(tensor=features[INPUT_TENSOR_NAME])
                          # Define operations
                          if mode in (Modes.PREDICT, Modes.EVAL):
                          logits = model(image, training=False)
                          predicted_indices = tf.argmax(input=logits, axis=1)
                          probabilities = tf.nn.softmax(logits, name='softmax_tensor')
                          if mode in (Modes.TRAIN):
                          logits = model(image, training=True)
                          global_step = tf.train.get_or_create_global_step()
                          loss = tf.losses.softmax_cross_entropy(
                          onehot_labels=labels, logits=logits)
                          tf.summary.scalar('OptimizeLoss', loss)
                          if mode in (Modes.EVAL):
                          logits = model(image, training=False)
                          global_step = tf.train.get_or_create_global_step()
                          loss = tf.losses.softmax_cross_entropy(
                          onehot_labels=labels, logits=logits)
                          tf.summary.scalar('OptimizeLoss', loss)
                          if mode == Modes.PREDICT:
                          predictions = {
                          'classes': predicted_indices,
                          'probabilities': probabilities
                          }
                          export_outputs = {
                          'predictions': tf.estimator.export.PredictOutput(predictions)
                          }
                          return tf.estimator.EstimatorSpec(
                          mode, predictions=predictions, export_outputs=export_outputs)
                          if mode == Modes.TRAIN:
                          optimizer = tf.train.AdamOptimizer(learning_rate=0.001)
                          train_op = optimizer.minimize(loss, global_step=global_step)
                          return tf.estimator.EstimatorSpec(mode, loss=loss, train_op=train_op)
                          if mode == Modes.EVAL:
                          eval_metric_ops = {
                          'accuracy': tf.metrics.accuracy(tf.argmax(labels, 1), predicted_indices)
                          }
                          return tf.estimator.EstimatorSpec(
                          mode, loss=loss, eval_metric_ops=eval_metric_ops)
                          def _input_fn(training_dir, input_shape, batch_size):
                          generator = tf.keras.preprocessing.image.ImageDataGenerator().flow_from_directory(training_dir, target_size=input_shape, batch_size=batch_size)
                          tensor_shapes = (tf.TensorShape([None, input_shape[0], input_shape[1], 3]), tf.TensorShape([None, NUM_CLASSES]))
                          tensor_types = (tf.float32, tf.float32)
                          dataset = tf.data.Dataset.from_generator(lambda: generator, tensor_types, tensor_shapes)
                          features, labels = dataset.make_one_shot_iterator().get_next()
                          return {INPUT_TENSOR_NAME: features}, labels
                          def train_input_fn(training_dir, hyperparameters):
                          return _input_fn(training_dir + '/train/', (224, 224), BATCH_SIZE)
                          def eval_input_fn(training_dir, hyperparameters):
                          return _input_fn(training_dir + '/test/', (224, 224), BATCH_SIZE)
                          def serving_input_fn(hyperparameters):
                          inputs = {INPUT_TENSOR_NAME: tf.placeholder(tf.float32, [None, 224, 224, 3])}
                          return tf.estimator.export.ServingInputReceiver(inputs, inputs)
                          

                          Activity

                          Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment

                          Metadata

                          Metadata

                          Assignees

                          No one assigned

                            Labels

                            No labels
                            No labels

                            Type

                            No type

                            Projects

                            No projects

                              Milestone

                              No milestone

                              Relationships

                              None yet

                              Development

                              No branches or pull requests

                              Issue actions

                              , 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
                              Skip to content

                              loading pre-trained weights for keras model is not supported in distributed training #264

                              Description

                              @WuyangLI

                              System Information

                              • Framework (e.g. TensorFlow) / Algorithm (e.g. KMeans): Tensorflow
                              • Framework Version: 1.8
                              • Python Version: 2
                              • CPU or GPU: GPU
                              • Python SDK Version: 1.5.1
                              • Are you using a custom image: No

                              Describe the problem

                              I created a distributed training job which trains a transfer learning model using VGG16.
                              The job would succeed if I don't load pre-trained weights when creating VGG16 backbone model.

                               backend = tf.keras.applications.vgg16.VGG16(weights=None, include_top=False,
                              input_shape=(224, 224, 3))
                              

                              However, exception would throw when I try to load pre-trained weights as done in the following code snippet:

                               backend = tf.keras.applications.vgg16.VGG16(weights='imagenet', include_top=False,
                              input_shape=(224, 224, 3))
                              

                              Note that, for non-distributed training, loading pre-trained weights would not cause any exception.

                              Minimal repro / logs

                              InvalidArgumentError (see above for traceback): Cannot assign a device for operation 'block5_conv3/bias': Operation was explicitly assigned to /job:ps/task:0 but available devices are [ /job:localhost/replica:0/task:0/device:CPU:0 ]. Make sure the device specification refers to a valid device.
                              #011 [[Node: block5_conv3/bias = VariableV2[_class=["loc:@block5_conv3/bias"], container="", dtype=DT_FLOAT, shape=[512], shared_name="", _device="/job:ps/task:0"]()]]
                              Traceback (most recent call last):
                              File "/usr/local/lib/python2.7/dist-packages/container_support/training.py", line 36, in start
                              fw.train()
                              File "/usr/local/lib/python2.7/dist-packages/tf_container/train_entry_point.py", line 164, in train
                              train_wrapper.train()
                              File "/usr/local/lib/python2.7/dist-packages/tf_container/trainer.py", line 73, in train
                              tf.estimator.train_and_evaluate(estimator=estimator, train_spec=train_spec, eval_spec=eval_spec)
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 439, in train_and_evaluate
                              executor.run()
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 546, in run
                              getattr(self, task_to_run)()
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 601, in run_master
                              self._start_distributed_training(saving_listeners=saving_listeners)
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/training.py", line 739, in _start_distributed_training
                              saving_listeners=saving_listeners)
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 363, in train
                              loss = self._train_model(input_fn, hooks, saving_listeners)
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 843, in _train_model
                              return self._train_model_default(input_fn, hooks, saving_listeners)
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 856, in _train_model_default
                              features, labels, model_fn_lib.ModeKeys.TRAIN, self.config)
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/estimator/estimator.py", line 831, in _call_model_fn
                              model_fn_results = self._model_fn(features=features, **kwargs)
                              File "/usr/local/lib/python2.7/dist-packages/tf_container/trainer.py", line 108, in _model_fn
                              return self.customer_script.model_fn(features, labels, mode, params)
                              File "/opt/ml/code/keras_distributed_transfer_learning.py", line 14, in model_fn
                              input_shape=(224, 224, 3))
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/applications/vgg16.py", line 225, in VGG16
                              model.load_weights(weights_path)
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/engine/network.py", line 1190, in load_weights
                              saving.load_weights_from_hdf5_group(f, self.layers)
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/engine/saving.py", line 719, in load_weights_from_hdf5_group
                              K.batch_set_value(weight_value_tuples)
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 2707, in batch_set_value
                              get_session().run(assign_ops, feed_dict=feed_dict)
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 442, in get_session
                              _initialize_variables(session)
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/keras/_impl/keras/backend.py", line 666, in _initialize_variables
                              [variables_module.is_variable_initialized(v) for v in candidate_vars])
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 900, in run
                              run_metadata_ptr)
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1135, in _run
                              feed_dict_tensor, options, run_metadata)
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1316, in _do_run
                              run_metadata)
                              File "/usr/local/lib/python2.7/dist-packages/tensorflow/python/client/session.py", line 1335, in _do_call
                              raise type(e)(node_def, op, message)
                              
                              • Exact command to reproduce:
                                code for creating the job
                              import sagemaker
                              from sagemaker.tensorflow import TensorFlow
                              from sagemaker.session import s3_input
                              from sagemaker import get_execution_role
                              sagemaker_session = sagemaker.Session()
                              role = get_execution_role()
                              training_steps = 100
                              evaluation_steps = 10
                              estimator = TensorFlow(
                              entry_point='keras_distributed_transfer_learning.py',
                              source_dir='./',
                              role=role,
                              training_steps=100,
                              evaluation_steps=10,
                              train_instance_count=2,
                              train_instance_type='ml.p2.xlarge',
                              input_mode='File')
                              input_dataset = s3_input('s3://xxxx/cats_and_dogs/')
                              estimator.fit(input_dataset)
                              

                              keras_distributed_transfer_learning.py

                              import tensorflow as tf
                              from tensorflow.python.estimator.model_fn import ModeKeys as Modes
                              INPUT_TENSOR_NAME = "input_1"
                              NUM_CLASSES = 2
                              BATCH_SIZE = 10
                              def model_fn(features, labels, mode, params):
                              """The model_fn argument for creating an Estimator."""
                              backend = tf.keras.applications.vgg16.VGG16(weights='imagenet', include_top=False,
                              input_shape=(224, 224, 3))
                              x = backend.output
                              x = tf.keras.layers.Flatten()(x)
                              x = tf.keras.layers.Dense(NUM_CLASSES, activation='softmax')(x)
                              model = tf.keras.models.Model(inputs=backend.input, outputs=x)
                              image = tf.keras.layers.Input(tensor=features[INPUT_TENSOR_NAME])
                              # Define operations
                              if mode in (Modes.PREDICT, Modes.EVAL):
                              logits = model(image, training=False)
                              predicted_indices = tf.argmax(input=logits, axis=1)
                              probabilities = tf.nn.softmax(logits, name='softmax_tensor')
                              if mode in (Modes.TRAIN):
                              logits = model(image, training=True)
                              global_step = tf.train.get_or_create_global_step()
                              loss = tf.losses.softmax_cross_entropy(
                              onehot_labels=labels, logits=logits)
                              tf.summary.scalar('OptimizeLoss', loss)
                              if mode in (Modes.EVAL):
                              logits = model(image, training=False)
                              global_step = tf.train.get_or_create_global_step()
                              loss = tf.losses.softmax_cross_entropy(
                              onehot_labels=labels, logits=logits)
                              tf.summary.scalar('OptimizeLoss', loss)
                              if mode == Modes.PREDICT:
                              predictions = {
                              'classes': predicted_indices,
                              'probabilities': probabilities
                              }
                              export_outputs = {
                              'predictions': tf.estimator.export.PredictOutput(predictions)
                              }
                              return tf.estimator.EstimatorSpec(
                              mode, predictions=predictions, export_outputs=export_outputs)
                              if mode == Modes.TRAIN:
                              optimizer = tf.train.AdamOptimizer(learning_rate=0.001)
                              train_op = optimizer.minimize(loss, global_step=global_step)
                              return tf.estimator.EstimatorSpec(mode, loss=loss, train_op=train_op)
                              if mode == Modes.EVAL:
                              eval_metric_ops = {
                              'accuracy': tf.metrics.accuracy(tf.argmax(labels, 1), predicted_indices)
                              }
                              return tf.estimator.EstimatorSpec(
                              mode, loss=loss, eval_metric_ops=eval_metric_ops)
                              def _input_fn(training_dir, input_shape, batch_size):
                              generator = tf.keras.preprocessing.image.ImageDataGenerator().flow_from_directory(training_dir, target_size=input_shape, batch_size=batch_size)
                              tensor_shapes = (tf.TensorShape([None, input_shape[0], input_shape[1], 3]), tf.TensorShape([None, NUM_CLASSES]))
                              tensor_types = (tf.float32, tf.float32)
                              dataset = tf.data.Dataset.from_generator(lambda: generator, tensor_types, tensor_shapes)
                              features, labels = dataset.make_one_shot_iterator().get_next()
                              return {INPUT_TENSOR_NAME: features}, labels
                              def train_input_fn(training_dir, hyperparameters):
                              return _input_fn(training_dir + '/train/', (224, 224), BATCH_SIZE)
                              def eval_input_fn(training_dir, hyperparameters):
                              return _input_fn(training_dir + '/test/', (224, 224), BATCH_SIZE)
                              def serving_input_fn(hyperparameters):
                              inputs = {INPUT_TENSOR_NAME: tf.placeholder(tf.float32, [None, 224, 224, 3])}
                              return tf.estimator.export.ServingInputReceiver(inputs, inputs)
                              

                              Activity

                              Sign up for free to join this conversation on GitHub. Already have an account? Sign in to comment

                              Metadata

                              Metadata

                              Assignees

                              No one assigned

                                Labels

                                No labels
                                No labels

                                Type

                                No type

                                Projects

                                No projects

                                  Milestone

                                  No milestone

                                  Relationships

                                  None yet

                                  Development

                                  No branches or pull requests

                                  Issue actions