Skip to content
Merged
1 change: 1 addition & 0 deletions docs/api/framework.rst
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,3 +9,4 @@ Framework-specific API
.. toctree::

pytorch
jax

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The main README should also mention JAX.
Grep for pytorch in that document and add JAX at those places and add a JAX example too as there is a PyTorch example.

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think README.md need more refactor to show all FWs supported by TE. Diretort adding JAX to all Pytorch appearance might mess up reading. We can submit a split PR to refactor README.md.

42 changes: 42 additions & 0 deletions docs/api/jax.rst
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
..
Copyright (c) 2022-2023, NVIDIA CORPORATION & AFFILIATES. All rights reserved.

See LICENSE for license information.

Jax
=======

.. autoapiclass:: transformer_engine.jax.MajorShardingType
.. autoapiclass:: transformer_engine.jax.ShardingType
.. autoapiclass:: transformer_engine.jax.TransformerLayerType


.. autoapiclass:: transformer_engine.jax.ShardingResource(dp_resource=None, tp_resource=None)


.. autoapiclass:: transformer_engine.jax.LayerNorm(epsilon=1e-6, layernorm_type='layernorm', **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.DenseGeneral(features, layernorm_type='layernorm', use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.LayerNormDenseGeneral(features, layernorm_type='layernorm', epsilon=1e-6, use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.LayerNormMLP(intermediate_dim=2048, layernorm_type='layernorm', epsilon=1e-6, use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.RelativePositionBiases(num_buckets, max_distance, num_heads, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.MultiHeadAttention(head_dim, num_heads, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.TransformerLayer(hidden_size=512, mlp_hidden_size=2048, num_attention_heads=8, **kwargs)
:members: __call__
Comment on lines +20 to +36

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

All of these modules are showing up in the Functions category below in the generated docs. @mingxu1067

@mingxu1067mingxu1067Mar 14, 2023

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Remove categories as PyTorch to make the style consistent.



.. autoapifunction:: transformer_engine.jax.extend_logical_axis_rules
.. autoapifunction:: transformer_engine.jax.fp8_autocast
.. autoapifunction:: transformer_engine.jax.update_collections
.. autoapifunction:: transformer_engine.jax.update_fp8_metas
8 changes: 5 additions & 3 deletions transformer_engine/jax/__init__.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,7 +3,9 @@
# See LICENSE for license information.
"""Transformer Engine bindings for JAX"""
from .fp8 import fp8_autocast, update_collections, update_fp8_metas
from .module import DenseGeneral, LayerNormDenseGeneral, LayerNormMLP, TransformerEngineBase
from .module import DenseGeneral, LayerNorm
from .module import LayerNormDenseGeneral, LayerNormMLP, TransformerEngineBase
from .transformer import extend_logical_axis_rules
from .transformer import RelativePositionBiases, TransformerLayer, TransformerLayerType
from .sharding import ShardingResource
from .transformer import MultiHeadAttention, RelativePositionBiases
from .transformer import TransformerLayer, TransformerLayerType
from .sharding import MajorShardingType, ShardingResource, ShardingType
2 changes: 1 addition & 1 deletion transformer_engine/jax/csrc/extensions.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -53,7 +53,7 @@ PYBIND11_MODULE(transformer_engine_jax, m) {
m.def("pack_norm_descriptor", &PackCustomCallNormDescriptor);
m.def("pack_softmax_descriptor", &PackCustomCallSoftmaxDescriptor);

pybind11::enum_<DType>(m, "DType")
pybind11::enum_<DType>(m, "DType", pybind11::module_local())
.value("kByte", DType::kByte)
.value("kInt32", DType::kInt32)
.value("kFloat32", DType::kFloat32)
Expand Down
33 changes: 17 additions & 16 deletions transformer_engine/jax/fp8.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -302,16 +302,16 @@ def fp8_autocast(enabled: bool = False,
.. note::
We only support :attr:`margin`, :attr:`fp8_format`, :attr:`interval` and
:attr:`amax_history_len` in recipe.DelayedScaling currently. Other parameters
in recipe.DelayedScaling would be ignored, even is set.
in recipe.DelayedScaling would be ignored, even if set.

Parameters
----------
enabled: bool, default = False
whether or not to enable fp8
Whether or not to enable fp8
fp8_recipe: recipe.DelayedScaling, default = None
recipe used for FP8 training.
sharding_resource: ShardingResource, defaule = None
specify the mesh axes for data and tensor parallelism to shard along.
Recipe used for FP8 training.
sharding_resource: ShardingResource, default = None
Specify the mesh axes for data and tensor parallelism to shard along.
If set to None, then ShardingResource() would be created.
"""
if fp8_recipe is None:
Expand All@@ -338,12 +338,11 @@ def fp8_autocast(enabled: bool = False,


# Function Wrappers
def update_collections(new: Collection, original: Collection) -> Collection:
def update_collections(new: Collection, original: Collection) -> FrozenDict:
r"""
A helper to update Flax's Collection. Collection is a union type of dict and
Flax's FrozenDict.
A helper to update Flax's Collection.

Collection = [dict, FrozenDict]
Collection = [dict, flax.core.frozen_dict.FrozenDict]

Parameters
----------
Expand All@@ -364,14 +363,16 @@ def update_fp8_metas(state: Collection) -> Collection:
r"""
Calculate new fp8 scales and its inverse via the followed formula

`exp` = floor(log2(`fp8_max` / `amax`)) - `margin`
`sf` = round(power(2, abs(exp)))
`sf` = `sf` if `amax` > 0.0, else original_scale
`sf` = `sf` if isfinite(`amax`), else original_scale)
`updated_scale` = `1/sf` if exp < 0, else `sf`
`updated_scale_inv` = `1/updated_scale`
.. code-block:: python

exp = floor(log2(fp8_max / amax)) - margin
sf = round(power(2, abs(exp)))
sf = sf if amax > 0.0, else original_scale
sf = sf if isfinite(amax), else original_scale)
updated_scale = 1/sf if exp < 0, else sf
updated_scale_inv = 1/updated_scale

Collection = [dict, FrozenDict]
Collection = [dict, flax.core.frozen_dict.FrozenDict]

Parameters
----------
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Add copy buttons to all
 blocks
(function() {
function addCopyButtons() {
document.querySelectorAll('pre code').forEach(function(codeBlock) {
if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;
codeBlock.parentElement.setAttribute('data-copy-added', 'true');
var btn = document.createElement('button');
btn.textContent = 'Copy';
btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';
btn.onmouseover = function() { this.style.opacity = '1'; };
btn.onmouseout = function() { this.style.opacity = '0.7'; };
btn.onclick = function() {
navigator.clipboard.writeText(codeBlock.textContent).then(function() {
btn.textContent = 'Copied!';
setTimeout(function() { btn.textContent = 'Copy'; }, 1500);
});
};
codeBlock.parentElement.style.position = 'relative';
codeBlock.parentElement.appendChild(btn);
});
}
addCopyButtons();
// Re-run on dynamic content
var observer = new MutationObserver(addCopyButtons);
observer.observe(document.body, { childList: true, subtree: true });
})();
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Adding documents to TE/JAX by mingxu1067 · Pull Request #87 · NVIDIA/TransformerEngine · GitHub
Skip to content
Merged
1 change: 1 addition & 0 deletions docs/api/framework.rst
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,3 +9,4 @@ Framework-specific API
.. toctree::

pytorch
jax

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The main README should also mention JAX.
Grep for pytorch in that document and add JAX at those places and add a JAX example too as there is a PyTorch example.

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think README.md need more refactor to show all FWs supported by TE. Diretort adding JAX to all Pytorch appearance might mess up reading. We can submit a split PR to refactor README.md.

42 changes: 42 additions & 0 deletions docs/api/jax.rst
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
..
Copyright (c) 2022-2023, NVIDIA CORPORATION & AFFILIATES. All rights reserved.

See LICENSE for license information.

Jax
=======

.. autoapiclass:: transformer_engine.jax.MajorShardingType
.. autoapiclass:: transformer_engine.jax.ShardingType
.. autoapiclass:: transformer_engine.jax.TransformerLayerType


.. autoapiclass:: transformer_engine.jax.ShardingResource(dp_resource=None, tp_resource=None)


.. autoapiclass:: transformer_engine.jax.LayerNorm(epsilon=1e-6, layernorm_type='layernorm', **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.DenseGeneral(features, layernorm_type='layernorm', use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.LayerNormDenseGeneral(features, layernorm_type='layernorm', epsilon=1e-6, use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.LayerNormMLP(intermediate_dim=2048, layernorm_type='layernorm', epsilon=1e-6, use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.RelativePositionBiases(num_buckets, max_distance, num_heads, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.MultiHeadAttention(head_dim, num_heads, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.TransformerLayer(hidden_size=512, mlp_hidden_size=2048, num_attention_heads=8, **kwargs)
:members: __call__
Comment on lines +20 to +36

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

All of these modules are showing up in the Functions category below in the generated docs. @mingxu1067

@mingxu1067mingxu1067Mar 14, 2023

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Remove categories as PyTorch to make the style consistent.



.. autoapifunction:: transformer_engine.jax.extend_logical_axis_rules
.. autoapifunction:: transformer_engine.jax.fp8_autocast
.. autoapifunction:: transformer_engine.jax.update_collections
.. autoapifunction:: transformer_engine.jax.update_fp8_metas
8 changes: 5 additions & 3 deletions transformer_engine/jax/__init__.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,7 +3,9 @@
# See LICENSE for license information.
"""Transformer Engine bindings for JAX"""
from .fp8 import fp8_autocast, update_collections, update_fp8_metas
from .module import DenseGeneral, LayerNormDenseGeneral, LayerNormMLP, TransformerEngineBase
from .module import DenseGeneral, LayerNorm
from .module import LayerNormDenseGeneral, LayerNormMLP, TransformerEngineBase
from .transformer import extend_logical_axis_rules
from .transformer import RelativePositionBiases, TransformerLayer, TransformerLayerType
from .sharding import ShardingResource
from .transformer import MultiHeadAttention, RelativePositionBiases
from .transformer import TransformerLayer, TransformerLayerType
from .sharding import MajorShardingType, ShardingResource, ShardingType
2 changes: 1 addition & 1 deletion transformer_engine/jax/csrc/extensions.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -53,7 +53,7 @@ PYBIND11_MODULE(transformer_engine_jax, m) {
m.def("pack_norm_descriptor", &PackCustomCallNormDescriptor);
m.def("pack_softmax_descriptor", &PackCustomCallSoftmaxDescriptor);

pybind11::enum_<DType>(m, "DType")
pybind11::enum_<DType>(m, "DType", pybind11::module_local())
.value("kByte", DType::kByte)
.value("kInt32", DType::kInt32)
.value("kFloat32", DType::kFloat32)
Expand Down
33 changes: 17 additions & 16 deletions transformer_engine/jax/fp8.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -302,16 +302,16 @@ def fp8_autocast(enabled: bool = False,
.. note::
We only support :attr:`margin`, :attr:`fp8_format`, :attr:`interval` and
:attr:`amax_history_len` in recipe.DelayedScaling currently. Other parameters
in recipe.DelayedScaling would be ignored, even is set.
in recipe.DelayedScaling would be ignored, even if set.

Parameters
----------
enabled: bool, default = False
whether or not to enable fp8
Whether or not to enable fp8
fp8_recipe: recipe.DelayedScaling, default = None
recipe used for FP8 training.
sharding_resource: ShardingResource, defaule = None
specify the mesh axes for data and tensor parallelism to shard along.
Recipe used for FP8 training.
sharding_resource: ShardingResource, default = None
Specify the mesh axes for data and tensor parallelism to shard along.
If set to None, then ShardingResource() would be created.
"""
if fp8_recipe is None:
Expand All@@ -338,12 +338,11 @@ def fp8_autocast(enabled: bool = False,


# Function Wrappers
def update_collections(new: Collection, original: Collection) -> Collection:
def update_collections(new: Collection, original: Collection) -> FrozenDict:
r"""
A helper to update Flax's Collection. Collection is a union type of dict and
Flax's FrozenDict.
A helper to update Flax's Collection.

Collection = [dict, FrozenDict]
Collection = [dict, flax.core.frozen_dict.FrozenDict]

Parameters
----------
Expand All@@ -364,14 +363,16 @@ def update_fp8_metas(state: Collection) -> Collection:
r"""
Calculate new fp8 scales and its inverse via the followed formula

`exp` = floor(log2(`fp8_max` / `amax`)) - `margin`
`sf` = round(power(2, abs(exp)))
`sf` = `sf` if `amax` > 0.0, else original_scale
`sf` = `sf` if isfinite(`amax`), else original_scale)
`updated_scale` = `1/sf` if exp < 0, else `sf`
`updated_scale_inv` = `1/updated_scale`
.. code-block:: python

exp = floor(log2(fp8_max / amax)) - margin
sf = round(power(2, abs(exp)))
sf = sf if amax > 0.0, else original_scale
sf = sf if isfinite(amax), else original_scale)
updated_scale = 1/sf if exp < 0, else sf
updated_scale_inv = 1/updated_scale

Collection = [dict, FrozenDict]
Collection = [dict, flax.core.frozen_dict.FrozenDict]

Parameters
----------
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Force GitHub README to respect dark mode (function() { var style = document.createElement('style'); style.textContent = ' .markdown-body { color-scheme: dark light; } .markdown-body pre { background: #161b22 !important; } .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; } .markdown-body table th, .markdown-body table td { border-color: #30363d !important; } .markdown-body img { background: #0d1117; } .markdown-body blockquote { border-left-color: #8b949e; } .markdown-body hr { border-color: #30363d; } '; document.head.appendChild(style); })(); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' Adding documents to TE/JAX by mingxu1067 · Pull Request #87 · NVIDIA/TransformerEngine · GitHub
Skip to content
Merged
1 change: 1 addition & 0 deletions docs/api/framework.rst
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,3 +9,4 @@ Framework-specific API
.. toctree::

pytorch
jax

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The main README should also mention JAX.
Grep for pytorch in that document and add JAX at those places and add a JAX example too as there is a PyTorch example.

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think README.md need more refactor to show all FWs supported by TE. Diretort adding JAX to all Pytorch appearance might mess up reading. We can submit a split PR to refactor README.md.

42 changes: 42 additions & 0 deletions docs/api/jax.rst
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
..
Copyright (c) 2022-2023, NVIDIA CORPORATION & AFFILIATES. All rights reserved.

See LICENSE for license information.

Jax
=======

.. autoapiclass:: transformer_engine.jax.MajorShardingType
.. autoapiclass:: transformer_engine.jax.ShardingType
.. autoapiclass:: transformer_engine.jax.TransformerLayerType


.. autoapiclass:: transformer_engine.jax.ShardingResource(dp_resource=None, tp_resource=None)


.. autoapiclass:: transformer_engine.jax.LayerNorm(epsilon=1e-6, layernorm_type='layernorm', **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.DenseGeneral(features, layernorm_type='layernorm', use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.LayerNormDenseGeneral(features, layernorm_type='layernorm', epsilon=1e-6, use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.LayerNormMLP(intermediate_dim=2048, layernorm_type='layernorm', epsilon=1e-6, use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.RelativePositionBiases(num_buckets, max_distance, num_heads, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.MultiHeadAttention(head_dim, num_heads, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.TransformerLayer(hidden_size=512, mlp_hidden_size=2048, num_attention_heads=8, **kwargs)
:members: __call__
Comment on lines +20 to +36

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

All of these modules are showing up in the Functions category below in the generated docs. @mingxu1067

@mingxu1067mingxu1067Mar 14, 2023

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Remove categories as PyTorch to make the style consistent.



.. autoapifunction:: transformer_engine.jax.extend_logical_axis_rules
.. autoapifunction:: transformer_engine.jax.fp8_autocast
.. autoapifunction:: transformer_engine.jax.update_collections
.. autoapifunction:: transformer_engine.jax.update_fp8_metas
8 changes: 5 additions & 3 deletions transformer_engine/jax/__init__.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,7 +3,9 @@
# See LICENSE for license information.
"""Transformer Engine bindings for JAX"""
from .fp8 import fp8_autocast, update_collections, update_fp8_metas
from .module import DenseGeneral, LayerNormDenseGeneral, LayerNormMLP, TransformerEngineBase
from .module import DenseGeneral, LayerNorm
from .module import LayerNormDenseGeneral, LayerNormMLP, TransformerEngineBase
from .transformer import extend_logical_axis_rules
from .transformer import RelativePositionBiases, TransformerLayer, TransformerLayerType
from .sharding import ShardingResource
from .transformer import MultiHeadAttention, RelativePositionBiases
from .transformer import TransformerLayer, TransformerLayerType
from .sharding import MajorShardingType, ShardingResource, ShardingType
2 changes: 1 addition & 1 deletion transformer_engine/jax/csrc/extensions.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -53,7 +53,7 @@ PYBIND11_MODULE(transformer_engine_jax, m) {
m.def("pack_norm_descriptor", &PackCustomCallNormDescriptor);
m.def("pack_softmax_descriptor", &PackCustomCallSoftmaxDescriptor);

pybind11::enum_<DType>(m, "DType")
pybind11::enum_<DType>(m, "DType", pybind11::module_local())
.value("kByte", DType::kByte)
.value("kInt32", DType::kInt32)
.value("kFloat32", DType::kFloat32)
Expand Down
33 changes: 17 additions & 16 deletions transformer_engine/jax/fp8.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -302,16 +302,16 @@ def fp8_autocast(enabled: bool = False,
.. note::
We only support :attr:`margin`, :attr:`fp8_format`, :attr:`interval` and
:attr:`amax_history_len` in recipe.DelayedScaling currently. Other parameters
in recipe.DelayedScaling would be ignored, even is set.
in recipe.DelayedScaling would be ignored, even if set.

Parameters
----------
enabled: bool, default = False
whether or not to enable fp8
Whether or not to enable fp8
fp8_recipe: recipe.DelayedScaling, default = None
recipe used for FP8 training.
sharding_resource: ShardingResource, defaule = None
specify the mesh axes for data and tensor parallelism to shard along.
Recipe used for FP8 training.
sharding_resource: ShardingResource, default = None
Specify the mesh axes for data and tensor parallelism to shard along.
If set to None, then ShardingResource() would be created.
"""
if fp8_recipe is None:
Expand All@@ -338,12 +338,11 @@ def fp8_autocast(enabled: bool = False,


# Function Wrappers
def update_collections(new: Collection, original: Collection) -> Collection:
def update_collections(new: Collection, original: Collection) -> FrozenDict:
r"""
A helper to update Flax's Collection. Collection is a union type of dict and
Flax's FrozenDict.
A helper to update Flax's Collection.

Collection = [dict, FrozenDict]
Collection = [dict, flax.core.frozen_dict.FrozenDict]

Parameters
----------
Expand All@@ -364,14 +363,16 @@ def update_fp8_metas(state: Collection) -> Collection:
r"""
Calculate new fp8 scales and its inverse via the followed formula

`exp` = floor(log2(`fp8_max` / `amax`)) - `margin`
`sf` = round(power(2, abs(exp)))
`sf` = `sf` if `amax` > 0.0, else original_scale
`sf` = `sf` if isfinite(`amax`), else original_scale)
`updated_scale` = `1/sf` if exp < 0, else `sf`
`updated_scale_inv` = `1/updated_scale`
.. code-block:: python

exp = floor(log2(fp8_max / amax)) - margin
sf = round(power(2, abs(exp)))
sf = sf if amax > 0.0, else original_scale
sf = sf if isfinite(amax), else original_scale)
updated_scale = 1/sf if exp < 0, else sf
updated_scale_inv = 1/updated_scale

Collection = [dict, FrozenDict]
Collection = [dict, flax.core.frozen_dict.FrozenDict]

Parameters
----------
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Highlight search terms from Google/DuckDuckGo/Bing referrer (function() { var ref = document.referrer; var terms = []; if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) { var url = new URL(ref); var q = url.searchParams.get('q') || url.searchParams.get('p'); if (q) { terms = q.split(/\s+/).filter(function(t) { return t.length > 2; }); } } if (terms.length === 0) return; var style = document.createElement('style'); style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }'; document.head.appendChild(style); function highlight(node) { if (node.nodeType === 3) { // text node var text = node.textContent; var found = false; terms.forEach(function(term) { var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\]\\]/g, '\\') + ')', 'gi'); if (regex.test(text)) { found = true; var frag = document.createDocumentFragment(); var parts = text.split(regex); parts.forEach(function(part, i) { if (i % 2 === 0) { frag.appendChild(document.createTextNode(part)); } else { var span = document.createElement('span'); span.className = 'userscript-highlight'; span.textContent = part; frag.appendChild(span); } }); node.parentNode.replaceChild(frag, node); } }); } else if (node.nodeType === 1 && node.childNodes) { // element var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT']; if (!skipTags.includes(node.tagName)) { Array.from(node.childNodes).forEach(highlight); } } } highlight(document.body); // Re-highlight on dynamic content var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1 || node.nodeType === 3) highlight(node); }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' Adding documents to TE/JAX by mingxu1067 · Pull Request #87 · NVIDIA/TransformerEngine · GitHub
Skip to content
Merged
1 change: 1 addition & 0 deletions docs/api/framework.rst
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,3 +9,4 @@ Framework-specific API
.. toctree::

pytorch
jax

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The main README should also mention JAX.
Grep for pytorch in that document and add JAX at those places and add a JAX example too as there is a PyTorch example.

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think README.md need more refactor to show all FWs supported by TE. Diretort adding JAX to all Pytorch appearance might mess up reading. We can submit a split PR to refactor README.md.

42 changes: 42 additions & 0 deletions docs/api/jax.rst
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
..
Copyright (c) 2022-2023, NVIDIA CORPORATION & AFFILIATES. All rights reserved.

See LICENSE for license information.

Jax
=======

.. autoapiclass:: transformer_engine.jax.MajorShardingType
.. autoapiclass:: transformer_engine.jax.ShardingType
.. autoapiclass:: transformer_engine.jax.TransformerLayerType


.. autoapiclass:: transformer_engine.jax.ShardingResource(dp_resource=None, tp_resource=None)


.. autoapiclass:: transformer_engine.jax.LayerNorm(epsilon=1e-6, layernorm_type='layernorm', **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.DenseGeneral(features, layernorm_type='layernorm', use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.LayerNormDenseGeneral(features, layernorm_type='layernorm', epsilon=1e-6, use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.LayerNormMLP(intermediate_dim=2048, layernorm_type='layernorm', epsilon=1e-6, use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.RelativePositionBiases(num_buckets, max_distance, num_heads, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.MultiHeadAttention(head_dim, num_heads, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.TransformerLayer(hidden_size=512, mlp_hidden_size=2048, num_attention_heads=8, **kwargs)
:members: __call__
Comment on lines +20 to +36

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

All of these modules are showing up in the Functions category below in the generated docs. @mingxu1067

@mingxu1067mingxu1067Mar 14, 2023

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Remove categories as PyTorch to make the style consistent.



.. autoapifunction:: transformer_engine.jax.extend_logical_axis_rules
.. autoapifunction:: transformer_engine.jax.fp8_autocast
.. autoapifunction:: transformer_engine.jax.update_collections
.. autoapifunction:: transformer_engine.jax.update_fp8_metas
8 changes: 5 additions & 3 deletions transformer_engine/jax/__init__.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,7 +3,9 @@
# See LICENSE for license information.
"""Transformer Engine bindings for JAX"""
from .fp8 import fp8_autocast, update_collections, update_fp8_metas
from .module import DenseGeneral, LayerNormDenseGeneral, LayerNormMLP, TransformerEngineBase
from .module import DenseGeneral, LayerNorm
from .module import LayerNormDenseGeneral, LayerNormMLP, TransformerEngineBase
from .transformer import extend_logical_axis_rules
from .transformer import RelativePositionBiases, TransformerLayer, TransformerLayerType
from .sharding import ShardingResource
from .transformer import MultiHeadAttention, RelativePositionBiases
from .transformer import TransformerLayer, TransformerLayerType
from .sharding import MajorShardingType, ShardingResource, ShardingType
2 changes: 1 addition & 1 deletion transformer_engine/jax/csrc/extensions.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -53,7 +53,7 @@ PYBIND11_MODULE(transformer_engine_jax, m) {
m.def("pack_norm_descriptor", &PackCustomCallNormDescriptor);
m.def("pack_softmax_descriptor", &PackCustomCallSoftmaxDescriptor);

pybind11::enum_<DType>(m, "DType")
pybind11::enum_<DType>(m, "DType", pybind11::module_local())
.value("kByte", DType::kByte)
.value("kInt32", DType::kInt32)
.value("kFloat32", DType::kFloat32)
Expand Down
33 changes: 17 additions & 16 deletions transformer_engine/jax/fp8.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -302,16 +302,16 @@ def fp8_autocast(enabled: bool = False,
.. note::
We only support :attr:`margin`, :attr:`fp8_format`, :attr:`interval` and
:attr:`amax_history_len` in recipe.DelayedScaling currently. Other parameters
in recipe.DelayedScaling would be ignored, even is set.
in recipe.DelayedScaling would be ignored, even if set.

Parameters
----------
enabled: bool, default = False
whether or not to enable fp8
Whether or not to enable fp8
fp8_recipe: recipe.DelayedScaling, default = None
recipe used for FP8 training.
sharding_resource: ShardingResource, defaule = None
specify the mesh axes for data and tensor parallelism to shard along.
Recipe used for FP8 training.
sharding_resource: ShardingResource, default = None
Specify the mesh axes for data and tensor parallelism to shard along.
If set to None, then ShardingResource() would be created.
"""
if fp8_recipe is None:
Expand All@@ -338,12 +338,11 @@ def fp8_autocast(enabled: bool = False,


# Function Wrappers
def update_collections(new: Collection, original: Collection) -> Collection:
def update_collections(new: Collection, original: Collection) -> FrozenDict:
r"""
A helper to update Flax's Collection. Collection is a union type of dict and
Flax's FrozenDict.
A helper to update Flax's Collection.

Collection = [dict, FrozenDict]
Collection = [dict, flax.core.frozen_dict.FrozenDict]

Parameters
----------
Expand All@@ -364,14 +363,16 @@ def update_fp8_metas(state: Collection) -> Collection:
r"""
Calculate new fp8 scales and its inverse via the followed formula

`exp` = floor(log2(`fp8_max` / `amax`)) - `margin`
`sf` = round(power(2, abs(exp)))
`sf` = `sf` if `amax` > 0.0, else original_scale
`sf` = `sf` if isfinite(`amax`), else original_scale)
`updated_scale` = `1/sf` if exp < 0, else `sf`
`updated_scale_inv` = `1/updated_scale`
.. code-block:: python

exp = floor(log2(fp8_max / amax)) - margin
sf = round(power(2, abs(exp)))
sf = sf if amax > 0.0, else original_scale
sf = sf if isfinite(amax), else original_scale)
updated_scale = 1/sf if exp < 0, else sf
updated_scale_inv = 1/updated_scale

Collection = [dict, FrozenDict]
Collection = [dict, flax.core.frozen_dict.FrozenDict]

Parameters
----------
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Strip utm_, fbclid, gclid, etc. from all links on page (function() { var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content', 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid', 'ref', 'ref_src', 'source', 'medium', 'campaign']; function cleanUrl(url) { try { var u = new URL(url, window.location.origin); var changed = false; trackingParams.forEach(function(p) { if (u.searchParams.has(p)) { u.searchParams.delete(p); changed = true; } }); return changed ? u.toString() : url; } catch (e) { return url; } } function cleanLinks() { document.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } cleanLinks(); var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1) { if (node.tagName === 'A') cleanLinks(); node.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + ' Adding documents to TE/JAX by mingxu1067 · Pull Request #87 · NVIDIA/TransformerEngine · GitHub
Skip to content
Merged
1 change: 1 addition & 0 deletions docs/api/framework.rst
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,3 +9,4 @@ Framework-specific API
.. toctree::

pytorch
jax

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The main README should also mention JAX.
Grep for pytorch in that document and add JAX at those places and add a JAX example too as there is a PyTorch example.

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think README.md need more refactor to show all FWs supported by TE. Diretort adding JAX to all Pytorch appearance might mess up reading. We can submit a split PR to refactor README.md.

42 changes: 42 additions & 0 deletions docs/api/jax.rst
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
..
Copyright (c) 2022-2023, NVIDIA CORPORATION & AFFILIATES. All rights reserved.

See LICENSE for license information.

Jax
=======

.. autoapiclass:: transformer_engine.jax.MajorShardingType
.. autoapiclass:: transformer_engine.jax.ShardingType
.. autoapiclass:: transformer_engine.jax.TransformerLayerType


.. autoapiclass:: transformer_engine.jax.ShardingResource(dp_resource=None, tp_resource=None)


.. autoapiclass:: transformer_engine.jax.LayerNorm(epsilon=1e-6, layernorm_type='layernorm', **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.DenseGeneral(features, layernorm_type='layernorm', use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.LayerNormDenseGeneral(features, layernorm_type='layernorm', epsilon=1e-6, use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.LayerNormMLP(intermediate_dim=2048, layernorm_type='layernorm', epsilon=1e-6, use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.RelativePositionBiases(num_buckets, max_distance, num_heads, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.MultiHeadAttention(head_dim, num_heads, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.TransformerLayer(hidden_size=512, mlp_hidden_size=2048, num_attention_heads=8, **kwargs)
:members: __call__
Comment on lines +20 to +36

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

All of these modules are showing up in the Functions category below in the generated docs. @mingxu1067

@mingxu1067mingxu1067Mar 14, 2023

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Remove categories as PyTorch to make the style consistent.



.. autoapifunction:: transformer_engine.jax.extend_logical_axis_rules
.. autoapifunction:: transformer_engine.jax.fp8_autocast
.. autoapifunction:: transformer_engine.jax.update_collections
.. autoapifunction:: transformer_engine.jax.update_fp8_metas
8 changes: 5 additions & 3 deletions transformer_engine/jax/__init__.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,7 +3,9 @@
# See LICENSE for license information.
"""Transformer Engine bindings for JAX"""
from .fp8 import fp8_autocast, update_collections, update_fp8_metas
from .module import DenseGeneral, LayerNormDenseGeneral, LayerNormMLP, TransformerEngineBase
from .module import DenseGeneral, LayerNorm
from .module import LayerNormDenseGeneral, LayerNormMLP, TransformerEngineBase
from .transformer import extend_logical_axis_rules
from .transformer import RelativePositionBiases, TransformerLayer, TransformerLayerType
from .sharding import ShardingResource
from .transformer import MultiHeadAttention, RelativePositionBiases
from .transformer import TransformerLayer, TransformerLayerType
from .sharding import MajorShardingType, ShardingResource, ShardingType
2 changes: 1 addition & 1 deletion transformer_engine/jax/csrc/extensions.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -53,7 +53,7 @@ PYBIND11_MODULE(transformer_engine_jax, m) {
m.def("pack_norm_descriptor", &PackCustomCallNormDescriptor);
m.def("pack_softmax_descriptor", &PackCustomCallSoftmaxDescriptor);

pybind11::enum_<DType>(m, "DType")
pybind11::enum_<DType>(m, "DType", pybind11::module_local())
.value("kByte", DType::kByte)
.value("kInt32", DType::kInt32)
.value("kFloat32", DType::kFloat32)
Expand Down
33 changes: 17 additions & 16 deletions transformer_engine/jax/fp8.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -302,16 +302,16 @@ def fp8_autocast(enabled: bool = False,
.. note::
We only support :attr:`margin`, :attr:`fp8_format`, :attr:`interval` and
:attr:`amax_history_len` in recipe.DelayedScaling currently. Other parameters
in recipe.DelayedScaling would be ignored, even is set.
in recipe.DelayedScaling would be ignored, even if set.

Parameters
----------
enabled: bool, default = False
whether or not to enable fp8
Whether or not to enable fp8
fp8_recipe: recipe.DelayedScaling, default = None
recipe used for FP8 training.
sharding_resource: ShardingResource, defaule = None
specify the mesh axes for data and tensor parallelism to shard along.
Recipe used for FP8 training.
sharding_resource: ShardingResource, default = None
Specify the mesh axes for data and tensor parallelism to shard along.
If set to None, then ShardingResource() would be created.
"""
if fp8_recipe is None:
Expand All@@ -338,12 +338,11 @@ def fp8_autocast(enabled: bool = False,


# Function Wrappers
def update_collections(new: Collection, original: Collection) -> Collection:
def update_collections(new: Collection, original: Collection) -> FrozenDict:
r"""
A helper to update Flax's Collection. Collection is a union type of dict and
Flax's FrozenDict.
A helper to update Flax's Collection.

Collection = [dict, FrozenDict]
Collection = [dict, flax.core.frozen_dict.FrozenDict]

Parameters
----------
Expand All@@ -364,14 +363,16 @@ def update_fp8_metas(state: Collection) -> Collection:
r"""
Calculate new fp8 scales and its inverse via the followed formula

`exp` = floor(log2(`fp8_max` / `amax`)) - `margin`
`sf` = round(power(2, abs(exp)))
`sf` = `sf` if `amax` > 0.0, else original_scale
`sf` = `sf` if isfinite(`amax`), else original_scale)
`updated_scale` = `1/sf` if exp < 0, else `sf`
`updated_scale_inv` = `1/updated_scale`
.. code-block:: python

exp = floor(log2(fp8_max / amax)) - margin
sf = round(power(2, abs(exp)))
sf = sf if amax > 0.0, else original_scale
sf = sf if isfinite(amax), else original_scale)
updated_scale = 1/sf if exp < 0, else sf
updated_scale_inv = 1/updated_scale

Collection = [dict, FrozenDict]
Collection = [dict, flax.core.frozen_dict.FrozenDict]

Parameters
----------
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Auto-enable theater mode on YouTube (function() { function tryTheater() { var btn = document.querySelector('button[aria-label="Theater mode"], ytd-player #player button[title="Theater mode"]'); if (btn && !btn.classList.contains('activated')) { btn.click(); } } // Try immediately tryTheater(); // Try after navigation (SPA) var lastUrl = location.href; setInterval(function() { if (location.href !== lastUrl) { lastUrl = location.href; setTimeout(tryTheater, 500); } }, 1000); // Also try on player load var observer = new MutationObserver(tryTheater); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' Adding documents to TE/JAX by mingxu1067 · Pull Request #87 · NVIDIA/TransformerEngine · GitHub
Skip to content
Merged
1 change: 1 addition & 0 deletions docs/api/framework.rst
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,3 +9,4 @@ Framework-specific API
.. toctree::

pytorch
jax

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The main README should also mention JAX.
Grep for pytorch in that document and add JAX at those places and add a JAX example too as there is a PyTorch example.

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think README.md need more refactor to show all FWs supported by TE. Diretort adding JAX to all Pytorch appearance might mess up reading. We can submit a split PR to refactor README.md.

42 changes: 42 additions & 0 deletions docs/api/jax.rst
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
..
Copyright (c) 2022-2023, NVIDIA CORPORATION & AFFILIATES. All rights reserved.

See LICENSE for license information.

Jax
=======

.. autoapiclass:: transformer_engine.jax.MajorShardingType
.. autoapiclass:: transformer_engine.jax.ShardingType
.. autoapiclass:: transformer_engine.jax.TransformerLayerType


.. autoapiclass:: transformer_engine.jax.ShardingResource(dp_resource=None, tp_resource=None)


.. autoapiclass:: transformer_engine.jax.LayerNorm(epsilon=1e-6, layernorm_type='layernorm', **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.DenseGeneral(features, layernorm_type='layernorm', use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.LayerNormDenseGeneral(features, layernorm_type='layernorm', epsilon=1e-6, use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.LayerNormMLP(intermediate_dim=2048, layernorm_type='layernorm', epsilon=1e-6, use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.RelativePositionBiases(num_buckets, max_distance, num_heads, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.MultiHeadAttention(head_dim, num_heads, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.TransformerLayer(hidden_size=512, mlp_hidden_size=2048, num_attention_heads=8, **kwargs)
:members: __call__
Comment on lines +20 to +36

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

All of these modules are showing up in the Functions category below in the generated docs. @mingxu1067

@mingxu1067mingxu1067Mar 14, 2023

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Remove categories as PyTorch to make the style consistent.



.. autoapifunction:: transformer_engine.jax.extend_logical_axis_rules
.. autoapifunction:: transformer_engine.jax.fp8_autocast
.. autoapifunction:: transformer_engine.jax.update_collections
.. autoapifunction:: transformer_engine.jax.update_fp8_metas
8 changes: 5 additions & 3 deletions transformer_engine/jax/__init__.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,7 +3,9 @@
# See LICENSE for license information.
"""Transformer Engine bindings for JAX"""
from .fp8 import fp8_autocast, update_collections, update_fp8_metas
from .module import DenseGeneral, LayerNormDenseGeneral, LayerNormMLP, TransformerEngineBase
from .module import DenseGeneral, LayerNorm
from .module import LayerNormDenseGeneral, LayerNormMLP, TransformerEngineBase
from .transformer import extend_logical_axis_rules
from .transformer import RelativePositionBiases, TransformerLayer, TransformerLayerType
from .sharding import ShardingResource
from .transformer import MultiHeadAttention, RelativePositionBiases
from .transformer import TransformerLayer, TransformerLayerType
from .sharding import MajorShardingType, ShardingResource, ShardingType
2 changes: 1 addition & 1 deletion transformer_engine/jax/csrc/extensions.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -53,7 +53,7 @@ PYBIND11_MODULE(transformer_engine_jax, m) {
m.def("pack_norm_descriptor", &PackCustomCallNormDescriptor);
m.def("pack_softmax_descriptor", &PackCustomCallSoftmaxDescriptor);

pybind11::enum_<DType>(m, "DType")
pybind11::enum_<DType>(m, "DType", pybind11::module_local())
.value("kByte", DType::kByte)
.value("kInt32", DType::kInt32)
.value("kFloat32", DType::kFloat32)
Expand Down
33 changes: 17 additions & 16 deletions transformer_engine/jax/fp8.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -302,16 +302,16 @@ def fp8_autocast(enabled: bool = False,
.. note::
We only support :attr:`margin`, :attr:`fp8_format`, :attr:`interval` and
:attr:`amax_history_len` in recipe.DelayedScaling currently. Other parameters
in recipe.DelayedScaling would be ignored, even is set.
in recipe.DelayedScaling would be ignored, even if set.

Parameters
----------
enabled: bool, default = False
whether or not to enable fp8
Whether or not to enable fp8
fp8_recipe: recipe.DelayedScaling, default = None
recipe used for FP8 training.
sharding_resource: ShardingResource, defaule = None
specify the mesh axes for data and tensor parallelism to shard along.
Recipe used for FP8 training.
sharding_resource: ShardingResource, default = None
Specify the mesh axes for data and tensor parallelism to shard along.
If set to None, then ShardingResource() would be created.
"""
if fp8_recipe is None:
Expand All@@ -338,12 +338,11 @@ def fp8_autocast(enabled: bool = False,


# Function Wrappers
def update_collections(new: Collection, original: Collection) -> Collection:
def update_collections(new: Collection, original: Collection) -> FrozenDict:
r"""
A helper to update Flax's Collection. Collection is a union type of dict and
Flax's FrozenDict.
A helper to update Flax's Collection.

Collection = [dict, FrozenDict]
Collection = [dict, flax.core.frozen_dict.FrozenDict]

Parameters
----------
Expand All@@ -364,14 +363,16 @@ def update_fp8_metas(state: Collection) -> Collection:
r"""
Calculate new fp8 scales and its inverse via the followed formula

`exp` = floor(log2(`fp8_max` / `amax`)) - `margin`
`sf` = round(power(2, abs(exp)))
`sf` = `sf` if `amax` > 0.0, else original_scale
`sf` = `sf` if isfinite(`amax`), else original_scale)
`updated_scale` = `1/sf` if exp < 0, else `sf`
`updated_scale_inv` = `1/updated_scale`
.. code-block:: python

exp = floor(log2(fp8_max / amax)) - margin
sf = round(power(2, abs(exp)))
sf = sf if amax > 0.0, else original_scale
sf = sf if isfinite(amax), else original_scale)
updated_scale = 1/sf if exp < 0, else sf
updated_scale_inv = 1/updated_scale

Collection = [dict, FrozenDict]
Collection = [dict, flax.core.frozen_dict.FrozenDict]

Parameters
----------
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Remove or un-stick sticky/fixed headers that block content (function() { function unstick() { document.querySelectorAll('header, nav, [role="banner"], .header, .navbar, .sticky, .fixed-top, [style*="position: fixed"], [style*="position:sticky"]').forEach(function(el) { if (el.style.position === 'fixed' || el.style.position === 'sticky' || getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') { el.style.position = 'static'; el.style.top = 'auto'; el.style.zIndex = 'auto'; } }); } unstick(); var observer = new MutationObserver(unstick); observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] }); })(); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' Adding documents to TE/JAX by mingxu1067 · Pull Request #87 · NVIDIA/TransformerEngine · GitHub
Skip to content
Merged
1 change: 1 addition & 0 deletions docs/api/framework.rst
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,3 +9,4 @@ Framework-specific API
.. toctree::

pytorch
jax

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The main README should also mention JAX.
Grep for pytorch in that document and add JAX at those places and add a JAX example too as there is a PyTorch example.

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think README.md need more refactor to show all FWs supported by TE. Diretort adding JAX to all Pytorch appearance might mess up reading. We can submit a split PR to refactor README.md.

42 changes: 42 additions & 0 deletions docs/api/jax.rst
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
..
Copyright (c) 2022-2023, NVIDIA CORPORATION & AFFILIATES. All rights reserved.

See LICENSE for license information.

Jax
=======

.. autoapiclass:: transformer_engine.jax.MajorShardingType
.. autoapiclass:: transformer_engine.jax.ShardingType
.. autoapiclass:: transformer_engine.jax.TransformerLayerType


.. autoapiclass:: transformer_engine.jax.ShardingResource(dp_resource=None, tp_resource=None)


.. autoapiclass:: transformer_engine.jax.LayerNorm(epsilon=1e-6, layernorm_type='layernorm', **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.DenseGeneral(features, layernorm_type='layernorm', use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.LayerNormDenseGeneral(features, layernorm_type='layernorm', epsilon=1e-6, use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.LayerNormMLP(intermediate_dim=2048, layernorm_type='layernorm', epsilon=1e-6, use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.RelativePositionBiases(num_buckets, max_distance, num_heads, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.MultiHeadAttention(head_dim, num_heads, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.TransformerLayer(hidden_size=512, mlp_hidden_size=2048, num_attention_heads=8, **kwargs)
:members: __call__
Comment on lines +20 to +36

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

All of these modules are showing up in the Functions category below in the generated docs. @mingxu1067

@mingxu1067mingxu1067Mar 14, 2023

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Remove categories as PyTorch to make the style consistent.



.. autoapifunction:: transformer_engine.jax.extend_logical_axis_rules
.. autoapifunction:: transformer_engine.jax.fp8_autocast
.. autoapifunction:: transformer_engine.jax.update_collections
.. autoapifunction:: transformer_engine.jax.update_fp8_metas
8 changes: 5 additions & 3 deletions transformer_engine/jax/__init__.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,7 +3,9 @@
# See LICENSE for license information.
"""Transformer Engine bindings for JAX"""
from .fp8 import fp8_autocast, update_collections, update_fp8_metas
from .module import DenseGeneral, LayerNormDenseGeneral, LayerNormMLP, TransformerEngineBase
from .module import DenseGeneral, LayerNorm
from .module import LayerNormDenseGeneral, LayerNormMLP, TransformerEngineBase
from .transformer import extend_logical_axis_rules
from .transformer import RelativePositionBiases, TransformerLayer, TransformerLayerType
from .sharding import ShardingResource
from .transformer import MultiHeadAttention, RelativePositionBiases
from .transformer import TransformerLayer, TransformerLayerType
from .sharding import MajorShardingType, ShardingResource, ShardingType
2 changes: 1 addition & 1 deletion transformer_engine/jax/csrc/extensions.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -53,7 +53,7 @@ PYBIND11_MODULE(transformer_engine_jax, m) {
m.def("pack_norm_descriptor", &PackCustomCallNormDescriptor);
m.def("pack_softmax_descriptor", &PackCustomCallSoftmaxDescriptor);

pybind11::enum_<DType>(m, "DType")
pybind11::enum_<DType>(m, "DType", pybind11::module_local())
.value("kByte", DType::kByte)
.value("kInt32", DType::kInt32)
.value("kFloat32", DType::kFloat32)
Expand Down
33 changes: 17 additions & 16 deletions transformer_engine/jax/fp8.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -302,16 +302,16 @@ def fp8_autocast(enabled: bool = False,
.. note::
We only support :attr:`margin`, :attr:`fp8_format`, :attr:`interval` and
:attr:`amax_history_len` in recipe.DelayedScaling currently. Other parameters
in recipe.DelayedScaling would be ignored, even is set.
in recipe.DelayedScaling would be ignored, even if set.

Parameters
----------
enabled: bool, default = False
whether or not to enable fp8
Whether or not to enable fp8
fp8_recipe: recipe.DelayedScaling, default = None
recipe used for FP8 training.
sharding_resource: ShardingResource, defaule = None
specify the mesh axes for data and tensor parallelism to shard along.
Recipe used for FP8 training.
sharding_resource: ShardingResource, default = None
Specify the mesh axes for data and tensor parallelism to shard along.
If set to None, then ShardingResource() would be created.
"""
if fp8_recipe is None:
Expand All@@ -338,12 +338,11 @@ def fp8_autocast(enabled: bool = False,


# Function Wrappers
def update_collections(new: Collection, original: Collection) -> Collection:
def update_collections(new: Collection, original: Collection) -> FrozenDict:
r"""
A helper to update Flax's Collection. Collection is a union type of dict and
Flax's FrozenDict.
A helper to update Flax's Collection.

Collection = [dict, FrozenDict]
Collection = [dict, flax.core.frozen_dict.FrozenDict]

Parameters
----------
Expand All@@ -364,14 +363,16 @@ def update_fp8_metas(state: Collection) -> Collection:
r"""
Calculate new fp8 scales and its inverse via the followed formula

`exp` = floor(log2(`fp8_max` / `amax`)) - `margin`
`sf` = round(power(2, abs(exp)))
`sf` = `sf` if `amax` > 0.0, else original_scale
`sf` = `sf` if isfinite(`amax`), else original_scale)
`updated_scale` = `1/sf` if exp < 0, else `sf`
`updated_scale_inv` = `1/updated_scale`
.. code-block:: python

exp = floor(log2(fp8_max / amax)) - margin
sf = round(power(2, abs(exp)))
sf = sf if amax > 0.0, else original_scale
sf = sf if isfinite(amax), else original_scale)
updated_scale = 1/sf if exp < 0, else sf
updated_scale_inv = 1/updated_scale

Collection = [dict, FrozenDict]
Collection = [dict, flax.core.frozen_dict.FrozenDict]

Parameters
----------
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Universal Dark Mode - works on any site (function() { var enabled = true; function applyDarkMode() { if (!enabled) return; // Create style element if it doesn't exist var style = document.getElementById('universal-dark-mode-style'); if (!style) { style = document.createElement('style'); style.id = 'universal-dark-mode-style'; document.head.appendChild(style); } // Dark mode CSS - inverts colors but preserves images/video style.textContent = ' /* Invert everything except media */ html { filter: invert(1) hue-rotate(180deg) !important; background: #1a1a2e !important; } /* Restore images, videos, iframes, canvas */ img, video, iframe, canvas, svg, picture, [style*="background-image"] { filter: invert(1) hue-rotate(180deg) !important; } /* Preserve specific elements that should not be inverted */ .no-dark-mode, .no-dark-mode *, [data-theme="light"], [data-theme="light"], .ace_editor, .ace_editor *, .CodeMirror, .CodeMirror *, .monaco-editor, .monaco-editor *, .markdown-body pre, .markdown-body pre *, .highlight, .highlight *, pre code, pre code * { filter: none !important; } /* Fix common UI elements */ .modal, .popup, .dropdown-menu, .tooltip, .popover { filter: invert(1) hue-rotate(180deg) !important; background: #2d2d44 !important; border-color: #444 !important; } /* Scrollbars */ ::-webkit-scrollbar { background: #1a1a2e !important; } ::-webkit-scrollbar-thumb { background: #444 !important; } ::-webkit-scrollbar-thumb:hover { background: #555 !important; } /* Selection */ ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; } ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; } '; } function removeDarkMode() { var style = document.getElementById('universal-dark-mode-style'); if (style) style.remove(); } // Toggle with Alt+Shift+D document.addEventListener('keydown', function(e) { if (e.altKey && e.shiftKey && e.key === 'D') { e.preventDefault(); enabled = !enabled; if (enabled) { applyDarkMode(); console.log('[Universal Dark Mode] Enabled'); } else { removeDarkMode(); console.log('[Universal Dark Mode] Disabled'); } } }); // Apply on load applyDarkMode(); // Re-apply on dynamic content var observer = new MutationObserver(function(mutations) { if (enabled && !document.getElementById('universal-dark-mode-style')) { applyDarkMode(); } }); observer.observe(document.head, { childList: true }); console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle'); })(); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })(); Adding documents to TE/JAX by mingxu1067 · Pull Request #87 · NVIDIA/TransformerEngine · GitHub
Skip to content
Merged
1 change: 1 addition & 0 deletions docs/api/framework.rst
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,3 +9,4 @@ Framework-specific API
.. toctree::

pytorch
jax

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The main README should also mention JAX.
Grep for pytorch in that document and add JAX at those places and add a JAX example too as there is a PyTorch example.

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think README.md need more refactor to show all FWs supported by TE. Diretort adding JAX to all Pytorch appearance might mess up reading. We can submit a split PR to refactor README.md.

42 changes: 42 additions & 0 deletions docs/api/jax.rst
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
..
Copyright (c) 2022-2023, NVIDIA CORPORATION & AFFILIATES. All rights reserved.

See LICENSE for license information.

Jax
=======

.. autoapiclass:: transformer_engine.jax.MajorShardingType
.. autoapiclass:: transformer_engine.jax.ShardingType
.. autoapiclass:: transformer_engine.jax.TransformerLayerType


.. autoapiclass:: transformer_engine.jax.ShardingResource(dp_resource=None, tp_resource=None)


.. autoapiclass:: transformer_engine.jax.LayerNorm(epsilon=1e-6, layernorm_type='layernorm', **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.DenseGeneral(features, layernorm_type='layernorm', use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.LayerNormDenseGeneral(features, layernorm_type='layernorm', epsilon=1e-6, use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.LayerNormMLP(intermediate_dim=2048, layernorm_type='layernorm', epsilon=1e-6, use_bias=False, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.RelativePositionBiases(num_buckets, max_distance, num_heads, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.MultiHeadAttention(head_dim, num_heads, **kwargs)
:members: __call__

.. autoapiclass:: transformer_engine.jax.TransformerLayer(hidden_size=512, mlp_hidden_size=2048, num_attention_heads=8, **kwargs)
:members: __call__
Comment on lines +20 to +36

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

All of these modules are showing up in the Functions category below in the generated docs. @mingxu1067

@mingxu1067mingxu1067Mar 14, 2023

Copy link
Copy Markdown
CollaboratorAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Remove categories as PyTorch to make the style consistent.



.. autoapifunction:: transformer_engine.jax.extend_logical_axis_rules
.. autoapifunction:: transformer_engine.jax.fp8_autocast
.. autoapifunction:: transformer_engine.jax.update_collections
.. autoapifunction:: transformer_engine.jax.update_fp8_metas
8 changes: 5 additions & 3 deletions transformer_engine/jax/__init__.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,7 +3,9 @@
# See LICENSE for license information.
"""Transformer Engine bindings for JAX"""
from .fp8 import fp8_autocast, update_collections, update_fp8_metas
from .module import DenseGeneral, LayerNormDenseGeneral, LayerNormMLP, TransformerEngineBase
from .module import DenseGeneral, LayerNorm
from .module import LayerNormDenseGeneral, LayerNormMLP, TransformerEngineBase
from .transformer import extend_logical_axis_rules
from .transformer import RelativePositionBiases, TransformerLayer, TransformerLayerType
from .sharding import ShardingResource
from .transformer import MultiHeadAttention, RelativePositionBiases
from .transformer import TransformerLayer, TransformerLayerType
from .sharding import MajorShardingType, ShardingResource, ShardingType
2 changes: 1 addition & 1 deletion transformer_engine/jax/csrc/extensions.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -53,7 +53,7 @@ PYBIND11_MODULE(transformer_engine_jax, m) {
m.def("pack_norm_descriptor", &PackCustomCallNormDescriptor);
m.def("pack_softmax_descriptor", &PackCustomCallSoftmaxDescriptor);

pybind11::enum_<DType>(m, "DType")
pybind11::enum_<DType>(m, "DType", pybind11::module_local())
.value("kByte", DType::kByte)
.value("kInt32", DType::kInt32)
.value("kFloat32", DType::kFloat32)
Expand Down
33 changes: 17 additions & 16 deletions transformer_engine/jax/fp8.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -302,16 +302,16 @@ def fp8_autocast(enabled: bool = False,
.. note::
We only support :attr:`margin`, :attr:`fp8_format`, :attr:`interval` and
:attr:`amax_history_len` in recipe.DelayedScaling currently. Other parameters
in recipe.DelayedScaling would be ignored, even is set.
in recipe.DelayedScaling would be ignored, even if set.

Parameters
----------
enabled: bool, default = False
whether or not to enable fp8
Whether or not to enable fp8
fp8_recipe: recipe.DelayedScaling, default = None
recipe used for FP8 training.
sharding_resource: ShardingResource, defaule = None
specify the mesh axes for data and tensor parallelism to shard along.
Recipe used for FP8 training.
sharding_resource: ShardingResource, default = None
Specify the mesh axes for data and tensor parallelism to shard along.
If set to None, then ShardingResource() would be created.
"""
if fp8_recipe is None:
Expand All@@ -338,12 +338,11 @@ def fp8_autocast(enabled: bool = False,


# Function Wrappers
def update_collections(new: Collection, original: Collection) -> Collection:
def update_collections(new: Collection, original: Collection) -> FrozenDict:
r"""
A helper to update Flax's Collection. Collection is a union type of dict and
Flax's FrozenDict.
A helper to update Flax's Collection.

Collection = [dict, FrozenDict]
Collection = [dict, flax.core.frozen_dict.FrozenDict]

Parameters
----------
Expand All@@ -364,14 +363,16 @@ def update_fp8_metas(state: Collection) -> Collection:
r"""
Calculate new fp8 scales and its inverse via the followed formula

`exp` = floor(log2(`fp8_max` / `amax`)) - `margin`
`sf` = round(power(2, abs(exp)))
`sf` = `sf` if `amax` > 0.0, else original_scale
`sf` = `sf` if isfinite(`amax`), else original_scale)
`updated_scale` = `1/sf` if exp < 0, else `sf`
`updated_scale_inv` = `1/updated_scale`
.. code-block:: python

exp = floor(log2(fp8_max / amax)) - margin
sf = round(power(2, abs(exp)))
sf = sf if amax > 0.0, else original_scale
sf = sf if isfinite(amax), else original_scale)
updated_scale = 1/sf if exp < 0, else sf
updated_scale_inv = 1/updated_scale

Collection = [dict, FrozenDict]
Collection = [dict, flax.core.frozen_dict.FrozenDict]

Parameters
----------
Expand Down
Loading