Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
172 changes: 171 additions & 1 deletion backends/mlx/examples/llm/export_llm_hf.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -346,6 +346,148 @@ def _export_with_custom_components(
_save_program(executorch_program, output_path)


def _export_with_offgraph_cache(
model_id: str,
revision: Optional[str],
output_path: str,
max_seq_len: int,
dtype: str,
qlinear: Optional[str],
qembedding: Optional[str],
no_tie_word_embeddings: bool = False,
qlinear_group_size: Optional[int] = None,
qembedding_group_size: Optional[int] = None,
) -> None:
"""
Export using the off-graph KV cache op (kvcache::update_and_attend).

Unlike the custom-components path, the cache is not a graph tensor: the model
is run with use_cache=False (no StaticCache wrapper, no HFStaticCache swap),
so each attention layer emits an update_and_attend node fed this step's k/v.
The cache is created and bound at runtime via a cache_key.
"""
import executorch.exir as exir
from executorch.backends.mlx import MLXPartitioner
from executorch.backends.mlx.llm.hf_attention import (
OffGraphExportWrapper,
register_mlx_offgraph_attention,
)
from executorch.backends.mlx.passes import get_default_passes
from executorch.exir import EdgeCompileConfig
from executorch.exir.capture._config import ExecutorchBackendConfig
from executorch.exir.passes import MemoryPlanningPass
from transformers import AutoModelForCausalLM

torch_dtype_map = {
"fp32": torch.float32,
"fp16": torch.float16,
"bf16": torch.bfloat16,
}
torch_dtype = torch_dtype_map.get(dtype, torch.bfloat16)

register_mlx_offgraph_attention()
logger.info("Registered MLX off-graph attention (update_and_attend)")

logger.info(f"Loading HuggingFace model: {model_id}")
load_kwargs = {
"torch_dtype": torch_dtype,
"low_cpu_mem_usage": True,
"attn_implementation": "mlx_offgraph",
}
if revision is not None:
load_kwargs["revision"] = revision
model = AutoModelForCausalLM.from_pretrained(model_id, **load_kwargs)
model.eval()

from executorch.backends.mlx.llm.quantization import quantize_model_

quantize_model_(
model,
qlinear_config=qlinear,
qlinear_group_size=qlinear_group_size,
qembedding_config=qembedding,
qembedding_group_size=qembedding_group_size,
tie_word_embeddings=getattr(model.config, "tie_word_embeddings", False)
and not no_tie_word_embeddings,
)

exportable = OffGraphExportWrapper(model)

# The cache is built before the graph runs, so the runtime cannot learn its
# shape from the model. Publish it, one entry per cache:
# n_caches how many caches to allocate
# kv_heads KV heads per cache
# head_dims head dim per cache -- gemma-4 mixes 256 and 512
# windows sliding window per cache, 0 = flat
from executorch.backends.mlx.llm.cache import resolve_hf_cache_layout

# resolve_hf_cache_layout drops the KV-shared tail, so these are indexed by
# cache -- the same indexing _cache_id maps layers onto.
layer_types, cache_kv_heads, cache_head_dims = resolve_hf_cache_layout(model.config)
text_config = model.config.get_text_config()
sliding_window = getattr(text_config, "sliding_window", None) or 0
cache_windows = [
sliding_window if t == "sliding_attention" else 0 for t in layer_types
]
kv_metadata = {
# The cache count, not num_hidden_layers: gemma-4 E2B shares KV across
# its tail, so 15 caches for 35 layers.
"get_n_caches": len(layer_types),
"get_kv_heads": torch.tensor(cache_kv_heads, dtype=torch.int32),
"get_head_dims": torch.tensor(cache_head_dims, dtype=torch.int32),
"get_windows": torch.tensor(cache_windows, dtype=torch.int32),
}
logger.info(
f"KV cache layout: {len(layer_types)} caches, "
f"{sum(1 for w in cache_windows if w)} sliding (window {sliding_window})"
)

logger.info("Exporting model with torch.export...")
seq_length = 3
example_input_ids = torch.zeros((1, seq_length), dtype=torch.long)
example_cache_position = torch.arange(seq_length, dtype=torch.long)

seq_len_dim = torch.export.Dim("seq_length_dim", max=max_seq_len - 1)
dynamic_shapes = {
"input_ids": {1: seq_len_dim},
"cache_position": {0: seq_len_dim},
}

with torch.no_grad():
exported_program = torch.export.export(
exportable,
args=(),
kwargs={
"input_ids": example_input_ids,
"cache_position": example_cache_position,
},
dynamic_shapes=dynamic_shapes,
strict=True,
)

logger.info("Delegating to MLX backend...")
edge_program = exir.to_edge_transform_and_lower(
{"forward": exported_program},
transform_passes=get_default_passes(),
constant_methods=kv_metadata,
partitioner=[MLXPartitioner()],
compile_config=EdgeCompileConfig(
_check_ir_validity=False,
_skip_dim_order=True,
),
)

logger.info("Exporting to ExecuTorch...")
executorch_program = edge_program.to_executorch(
config=ExecutorchBackendConfig(
extract_delegate_segments=True,
memory_planning_pass=MemoryPlanningPass(alloc_graph_input=True),
)
)

_save_program(executorch_program, output_path)


def _save_program(executorch_program, output_path: str) -> None:
"""Save the ExecuTorch program to disk."""
os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True)
Expand All@@ -366,6 +508,7 @@ def export_llama_hf(
qembedding: Optional[str] = None,
use_custom_sdpa: bool = False,
use_custom_kv_cache: bool = False,
use_offgraph_cache: bool = False,
no_tie_word_embeddings: bool = False,
qlinear_group_size: Optional[int] = None,
qembedding_group_size: Optional[int] = None,
Expand All@@ -383,7 +526,26 @@ def export_llama_hf(
use_custom_sdpa: Use MLX custom SDPA (mlx::custom_sdpa)
use_custom_kv_cache: Use MLX custom KV cache (mlx::kv_cache_update)
"""
if use_custom_sdpa or use_custom_kv_cache:
if use_offgraph_cache:
if use_custom_sdpa or use_custom_kv_cache:
raise ValueError(
"--use-offgraph-cache is exclusive with --use-custom-sdpa / "
"--use-custom-kv-cache (it replaces both)"
)
logger.info("Using off-graph KV cache (update_and_attend)")
_export_with_offgraph_cache(
model_id=model_id,
revision=revision,
output_path=output_path,
max_seq_len=max_seq_len,
dtype=dtype,
qlinear=qlinear,
qembedding=qembedding,
no_tie_word_embeddings=no_tie_word_embeddings,
qlinear_group_size=qlinear_group_size,
qembedding_group_size=qembedding_group_size,
)
elif use_custom_sdpa or use_custom_kv_cache:
logger.info(
f"Using custom components: sdpa={use_custom_sdpa}, "
f"kv_cache={use_custom_kv_cache}"
Expand DownExpand Up@@ -468,6 +630,13 @@ def main():
default=False,
help="Use MLX custom KV cache (mlx::kv_cache_update)",
)
parser.add_argument(
"--use-offgraph-cache",
action="store_true",
default=False,
help="Use the off-graph KV cache op (kvcache::update_and_attend); "
"replaces --use-custom-sdpa/--use-custom-kv-cache",
)

args = parser.parse_args()

Expand All@@ -481,6 +650,7 @@ def main():
qembedding=args.qembedding,
use_custom_sdpa=args.use_custom_sdpa,
use_custom_kv_cache=args.use_custom_kv_cache,
use_offgraph_cache=args.use_offgraph_cache,
no_tie_word_embeddings=args.no_tie_word_embeddings,
qlinear_group_size=args.qlinear_group_size,
qembedding_group_size=args.qembedding_group_size,
Expand Down
4 changes: 3 additions & 1 deletion backends/mlx/llm/cache.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -525,7 +525,9 @@ def update(
# Current HF ExecuTorch wrappers copy the requested cache position
# into each StaticCache layer's cumulative_length before forward().
if hasattr(self.layers[layer_idx], "cumulative_length"):
cache_position = self.layers[layer_idx].cumulative_length
# cumulative_length is a scalar; KVCache.update indexes [0], so
# give it the 1-D shape a cache_kwargs caller would have passed.
cache_position = self.layers[layer_idx].cumulative_length.reshape(1)
else:
raise RuntimeError(
"cache_position was not provided and the pinned "
Expand Down
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
172 changes: 171 additions & 1 deletion backends/mlx/examples/llm/export_llm_hf.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -346,6 +346,148 @@ def _export_with_custom_components(
_save_program(executorch_program, output_path)


def _export_with_offgraph_cache(
model_id: str,
revision: Optional[str],
output_path: str,
max_seq_len: int,
dtype: str,
qlinear: Optional[str],
qembedding: Optional[str],
no_tie_word_embeddings: bool = False,
qlinear_group_size: Optional[int] = None,
qembedding_group_size: Optional[int] = None,
) -> None:
"""
Export using the off-graph KV cache op (kvcache::update_and_attend).

Unlike the custom-components path, the cache is not a graph tensor: the model
is run with use_cache=False (no StaticCache wrapper, no HFStaticCache swap),
so each attention layer emits an update_and_attend node fed this step's k/v.
The cache is created and bound at runtime via a cache_key.
"""
import executorch.exir as exir
from executorch.backends.mlx import MLXPartitioner
from executorch.backends.mlx.llm.hf_attention import (
OffGraphExportWrapper,
register_mlx_offgraph_attention,
)
from executorch.backends.mlx.passes import get_default_passes
from executorch.exir import EdgeCompileConfig
from executorch.exir.capture._config import ExecutorchBackendConfig
from executorch.exir.passes import MemoryPlanningPass
from transformers import AutoModelForCausalLM

torch_dtype_map = {
"fp32": torch.float32,
"fp16": torch.float16,
"bf16": torch.bfloat16,
}
torch_dtype = torch_dtype_map.get(dtype, torch.bfloat16)

register_mlx_offgraph_attention()
logger.info("Registered MLX off-graph attention (update_and_attend)")

logger.info(f"Loading HuggingFace model: {model_id}")
load_kwargs = {
"torch_dtype": torch_dtype,
"low_cpu_mem_usage": True,
"attn_implementation": "mlx_offgraph",
}
if revision is not None:
load_kwargs["revision"] = revision
model = AutoModelForCausalLM.from_pretrained(model_id, **load_kwargs)
model.eval()

from executorch.backends.mlx.llm.quantization import quantize_model_

quantize_model_(
model,
qlinear_config=qlinear,
qlinear_group_size=qlinear_group_size,
qembedding_config=qembedding,
qembedding_group_size=qembedding_group_size,
tie_word_embeddings=getattr(model.config, "tie_word_embeddings", False)
and not no_tie_word_embeddings,
)

exportable = OffGraphExportWrapper(model)

# The cache is built before the graph runs, so the runtime cannot learn its
# shape from the model. Publish it, one entry per cache:
# n_caches how many caches to allocate
# kv_heads KV heads per cache
# head_dims head dim per cache -- gemma-4 mixes 256 and 512
# windows sliding window per cache, 0 = flat
from executorch.backends.mlx.llm.cache import resolve_hf_cache_layout

# resolve_hf_cache_layout drops the KV-shared tail, so these are indexed by
# cache -- the same indexing _cache_id maps layers onto.
layer_types, cache_kv_heads, cache_head_dims = resolve_hf_cache_layout(model.config)
text_config = model.config.get_text_config()
sliding_window = getattr(text_config, "sliding_window", None) or 0
cache_windows = [
sliding_window if t == "sliding_attention" else 0 for t in layer_types
]
kv_metadata = {
# The cache count, not num_hidden_layers: gemma-4 E2B shares KV across
# its tail, so 15 caches for 35 layers.
"get_n_caches": len(layer_types),
"get_kv_heads": torch.tensor(cache_kv_heads, dtype=torch.int32),
"get_head_dims": torch.tensor(cache_head_dims, dtype=torch.int32),
"get_windows": torch.tensor(cache_windows, dtype=torch.int32),
}
logger.info(
f"KV cache layout: {len(layer_types)} caches, "
f"{sum(1 for w in cache_windows if w)} sliding (window {sliding_window})"
)

logger.info("Exporting model with torch.export...")
seq_length = 3
example_input_ids = torch.zeros((1, seq_length), dtype=torch.long)
example_cache_position = torch.arange(seq_length, dtype=torch.long)

seq_len_dim = torch.export.Dim("seq_length_dim", max=max_seq_len - 1)
dynamic_shapes = {
"input_ids": {1: seq_len_dim},
"cache_position": {0: seq_len_dim},
}

with torch.no_grad():
exported_program = torch.export.export(
exportable,
args=(),
kwargs={
"input_ids": example_input_ids,
"cache_position": example_cache_position,
},
dynamic_shapes=dynamic_shapes,
strict=True,
)

logger.info("Delegating to MLX backend...")
edge_program = exir.to_edge_transform_and_lower(
{"forward": exported_program},
transform_passes=get_default_passes(),
constant_methods=kv_metadata,
partitioner=[MLXPartitioner()],
compile_config=EdgeCompileConfig(
_check_ir_validity=False,
_skip_dim_order=True,
),
)

logger.info("Exporting to ExecuTorch...")
executorch_program = edge_program.to_executorch(
config=ExecutorchBackendConfig(
extract_delegate_segments=True,
memory_planning_pass=MemoryPlanningPass(alloc_graph_input=True),
)
)

_save_program(executorch_program, output_path)


def _save_program(executorch_program, output_path: str) -> None:
"""Save the ExecuTorch program to disk."""
os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True)
Expand All@@ -366,6 +508,7 @@ def export_llama_hf(
qembedding: Optional[str] = None,
use_custom_sdpa: bool = False,
use_custom_kv_cache: bool = False,
use_offgraph_cache: bool = False,
no_tie_word_embeddings: bool = False,
qlinear_group_size: Optional[int] = None,
qembedding_group_size: Optional[int] = None,
Expand All@@ -383,7 +526,26 @@ def export_llama_hf(
use_custom_sdpa: Use MLX custom SDPA (mlx::custom_sdpa)
use_custom_kv_cache: Use MLX custom KV cache (mlx::kv_cache_update)
"""
if use_custom_sdpa or use_custom_kv_cache:
if use_offgraph_cache:
if use_custom_sdpa or use_custom_kv_cache:
raise ValueError(
"--use-offgraph-cache is exclusive with --use-custom-sdpa / "
"--use-custom-kv-cache (it replaces both)"
)
logger.info("Using off-graph KV cache (update_and_attend)")
_export_with_offgraph_cache(
model_id=model_id,
revision=revision,
output_path=output_path,
max_seq_len=max_seq_len,
dtype=dtype,
qlinear=qlinear,
qembedding=qembedding,
no_tie_word_embeddings=no_tie_word_embeddings,
qlinear_group_size=qlinear_group_size,
qembedding_group_size=qembedding_group_size,
)
elif use_custom_sdpa or use_custom_kv_cache:
logger.info(
f"Using custom components: sdpa={use_custom_sdpa}, "
f"kv_cache={use_custom_kv_cache}"
Expand DownExpand Up@@ -468,6 +630,13 @@ def main():
default=False,
help="Use MLX custom KV cache (mlx::kv_cache_update)",
)
parser.add_argument(
"--use-offgraph-cache",
action="store_true",
default=False,
help="Use the off-graph KV cache op (kvcache::update_and_attend); "
"replaces --use-custom-sdpa/--use-custom-kv-cache",
)

args = parser.parse_args()

Expand All@@ -481,6 +650,7 @@ def main():
qembedding=args.qembedding,
use_custom_sdpa=args.use_custom_sdpa,
use_custom_kv_cache=args.use_custom_kv_cache,
use_offgraph_cache=args.use_offgraph_cache,
no_tie_word_embeddings=args.no_tie_word_embeddings,
qlinear_group_size=args.qlinear_group_size,
qembedding_group_size=args.qembedding_group_size,
Expand Down
4 changes: 3 additions & 1 deletion backends/mlx/llm/cache.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -525,7 +525,9 @@ def update(
# Current HF ExecuTorch wrappers copy the requested cache position
# into each StaticCache layer's cumulative_length before forward().
if hasattr(self.layers[layer_idx], "cumulative_length"):
cache_position = self.layers[layer_idx].cumulative_length
# cumulative_length is a scalar; KVCache.update indexes [0], so
# give it the 1-D shape a cache_kwargs caller would have passed.
cache_position = self.layers[layer_idx].cumulative_length.reshape(1)
else:
raise RuntimeError(
"cache_position was not provided and the pinned "
Expand Down
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
172 changes: 171 additions & 1 deletion backends/mlx/examples/llm/export_llm_hf.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -346,6 +346,148 @@ def _export_with_custom_components(
_save_program(executorch_program, output_path)


def _export_with_offgraph_cache(
model_id: str,
revision: Optional[str],
output_path: str,
max_seq_len: int,
dtype: str,
qlinear: Optional[str],
qembedding: Optional[str],
no_tie_word_embeddings: bool = False,
qlinear_group_size: Optional[int] = None,
qembedding_group_size: Optional[int] = None,
) -> None:
"""
Export using the off-graph KV cache op (kvcache::update_and_attend).

Unlike the custom-components path, the cache is not a graph tensor: the model
is run with use_cache=False (no StaticCache wrapper, no HFStaticCache swap),
so each attention layer emits an update_and_attend node fed this step's k/v.
The cache is created and bound at runtime via a cache_key.
"""
import executorch.exir as exir
from executorch.backends.mlx import MLXPartitioner
from executorch.backends.mlx.llm.hf_attention import (
OffGraphExportWrapper,
register_mlx_offgraph_attention,
)
from executorch.backends.mlx.passes import get_default_passes
from executorch.exir import EdgeCompileConfig
from executorch.exir.capture._config import ExecutorchBackendConfig
from executorch.exir.passes import MemoryPlanningPass
from transformers import AutoModelForCausalLM

torch_dtype_map = {
"fp32": torch.float32,
"fp16": torch.float16,
"bf16": torch.bfloat16,
}
torch_dtype = torch_dtype_map.get(dtype, torch.bfloat16)

register_mlx_offgraph_attention()
logger.info("Registered MLX off-graph attention (update_and_attend)")

logger.info(f"Loading HuggingFace model: {model_id}")
load_kwargs = {
"torch_dtype": torch_dtype,
"low_cpu_mem_usage": True,
"attn_implementation": "mlx_offgraph",
}
if revision is not None:
load_kwargs["revision"] = revision
model = AutoModelForCausalLM.from_pretrained(model_id, **load_kwargs)
model.eval()

from executorch.backends.mlx.llm.quantization import quantize_model_

quantize_model_(
model,
qlinear_config=qlinear,
qlinear_group_size=qlinear_group_size,
qembedding_config=qembedding,
qembedding_group_size=qembedding_group_size,
tie_word_embeddings=getattr(model.config, "tie_word_embeddings", False)
and not no_tie_word_embeddings,
)

exportable = OffGraphExportWrapper(model)

# The cache is built before the graph runs, so the runtime cannot learn its
# shape from the model. Publish it, one entry per cache:
# n_caches how many caches to allocate
# kv_heads KV heads per cache
# head_dims head dim per cache -- gemma-4 mixes 256 and 512
# windows sliding window per cache, 0 = flat
from executorch.backends.mlx.llm.cache import resolve_hf_cache_layout

# resolve_hf_cache_layout drops the KV-shared tail, so these are indexed by
# cache -- the same indexing _cache_id maps layers onto.
layer_types, cache_kv_heads, cache_head_dims = resolve_hf_cache_layout(model.config)
text_config = model.config.get_text_config()
sliding_window = getattr(text_config, "sliding_window", None) or 0
cache_windows = [
sliding_window if t == "sliding_attention" else 0 for t in layer_types
]
kv_metadata = {
# The cache count, not num_hidden_layers: gemma-4 E2B shares KV across
# its tail, so 15 caches for 35 layers.
"get_n_caches": len(layer_types),
"get_kv_heads": torch.tensor(cache_kv_heads, dtype=torch.int32),
"get_head_dims": torch.tensor(cache_head_dims, dtype=torch.int32),
"get_windows": torch.tensor(cache_windows, dtype=torch.int32),
}
logger.info(
f"KV cache layout: {len(layer_types)} caches, "
f"{sum(1 for w in cache_windows if w)} sliding (window {sliding_window})"
)

logger.info("Exporting model with torch.export...")
seq_length = 3
example_input_ids = torch.zeros((1, seq_length), dtype=torch.long)
example_cache_position = torch.arange(seq_length, dtype=torch.long)

seq_len_dim = torch.export.Dim("seq_length_dim", max=max_seq_len - 1)
dynamic_shapes = {
"input_ids": {1: seq_len_dim},
"cache_position": {0: seq_len_dim},
}

with torch.no_grad():
exported_program = torch.export.export(
exportable,
args=(),
kwargs={
"input_ids": example_input_ids,
"cache_position": example_cache_position,
},
dynamic_shapes=dynamic_shapes,
strict=True,
)

logger.info("Delegating to MLX backend...")
edge_program = exir.to_edge_transform_and_lower(
{"forward": exported_program},
transform_passes=get_default_passes(),
constant_methods=kv_metadata,
partitioner=[MLXPartitioner()],
compile_config=EdgeCompileConfig(
_check_ir_validity=False,
_skip_dim_order=True,
),
)

logger.info("Exporting to ExecuTorch...")
executorch_program = edge_program.to_executorch(
config=ExecutorchBackendConfig(
extract_delegate_segments=True,
memory_planning_pass=MemoryPlanningPass(alloc_graph_input=True),
)
)

_save_program(executorch_program, output_path)


def _save_program(executorch_program, output_path: str) -> None:
"""Save the ExecuTorch program to disk."""
os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True)
Expand All@@ -366,6 +508,7 @@ def export_llama_hf(
qembedding: Optional[str] = None,
use_custom_sdpa: bool = False,
use_custom_kv_cache: bool = False,
use_offgraph_cache: bool = False,
no_tie_word_embeddings: bool = False,
qlinear_group_size: Optional[int] = None,
qembedding_group_size: Optional[int] = None,
Expand All@@ -383,7 +526,26 @@ def export_llama_hf(
use_custom_sdpa: Use MLX custom SDPA (mlx::custom_sdpa)
use_custom_kv_cache: Use MLX custom KV cache (mlx::kv_cache_update)
"""
if use_custom_sdpa or use_custom_kv_cache:
if use_offgraph_cache:
if use_custom_sdpa or use_custom_kv_cache:
raise ValueError(
"--use-offgraph-cache is exclusive with --use-custom-sdpa / "
"--use-custom-kv-cache (it replaces both)"
)
logger.info("Using off-graph KV cache (update_and_attend)")
_export_with_offgraph_cache(
model_id=model_id,
revision=revision,
output_path=output_path,
max_seq_len=max_seq_len,
dtype=dtype,
qlinear=qlinear,
qembedding=qembedding,
no_tie_word_embeddings=no_tie_word_embeddings,
qlinear_group_size=qlinear_group_size,
qembedding_group_size=qembedding_group_size,
)
elif use_custom_sdpa or use_custom_kv_cache:
logger.info(
f"Using custom components: sdpa={use_custom_sdpa}, "
f"kv_cache={use_custom_kv_cache}"
Expand DownExpand Up@@ -468,6 +630,13 @@ def main():
default=False,
help="Use MLX custom KV cache (mlx::kv_cache_update)",
)
parser.add_argument(
"--use-offgraph-cache",
action="store_true",
default=False,
help="Use the off-graph KV cache op (kvcache::update_and_attend); "
"replaces --use-custom-sdpa/--use-custom-kv-cache",
)

args = parser.parse_args()

Expand All@@ -481,6 +650,7 @@ def main():
qembedding=args.qembedding,
use_custom_sdpa=args.use_custom_sdpa,
use_custom_kv_cache=args.use_custom_kv_cache,
use_offgraph_cache=args.use_offgraph_cache,
no_tie_word_embeddings=args.no_tie_word_embeddings,
qlinear_group_size=args.qlinear_group_size,
qembedding_group_size=args.qembedding_group_size,
Expand Down
4 changes: 3 additions & 1 deletion backends/mlx/llm/cache.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -525,7 +525,9 @@ def update(
# Current HF ExecuTorch wrappers copy the requested cache position
# into each StaticCache layer's cumulative_length before forward().
if hasattr(self.layers[layer_idx], "cumulative_length"):
cache_position = self.layers[layer_idx].cumulative_length
# cumulative_length is a scalar; KVCache.update indexes [0], so
# give it the 1-D shape a cache_kwargs caller would have passed.
cache_position = self.layers[layer_idx].cumulative_length.reshape(1)
else:
raise RuntimeError(
"cache_position was not provided and the pinned "
Expand Down
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
172 changes: 171 additions & 1 deletion backends/mlx/examples/llm/export_llm_hf.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -346,6 +346,148 @@ def _export_with_custom_components(
_save_program(executorch_program, output_path)


def _export_with_offgraph_cache(
model_id: str,
revision: Optional[str],
output_path: str,
max_seq_len: int,
dtype: str,
qlinear: Optional[str],
qembedding: Optional[str],
no_tie_word_embeddings: bool = False,
qlinear_group_size: Optional[int] = None,
qembedding_group_size: Optional[int] = None,
) -> None:
"""
Export using the off-graph KV cache op (kvcache::update_and_attend).

Unlike the custom-components path, the cache is not a graph tensor: the model
is run with use_cache=False (no StaticCache wrapper, no HFStaticCache swap),
so each attention layer emits an update_and_attend node fed this step's k/v.
The cache is created and bound at runtime via a cache_key.
"""
import executorch.exir as exir
from executorch.backends.mlx import MLXPartitioner
from executorch.backends.mlx.llm.hf_attention import (
OffGraphExportWrapper,
register_mlx_offgraph_attention,
)
from executorch.backends.mlx.passes import get_default_passes
from executorch.exir import EdgeCompileConfig
from executorch.exir.capture._config import ExecutorchBackendConfig
from executorch.exir.passes import MemoryPlanningPass
from transformers import AutoModelForCausalLM

torch_dtype_map = {
"fp32": torch.float32,
"fp16": torch.float16,
"bf16": torch.bfloat16,
}
torch_dtype = torch_dtype_map.get(dtype, torch.bfloat16)

register_mlx_offgraph_attention()
logger.info("Registered MLX off-graph attention (update_and_attend)")

logger.info(f"Loading HuggingFace model: {model_id}")
load_kwargs = {
"torch_dtype": torch_dtype,
"low_cpu_mem_usage": True,
"attn_implementation": "mlx_offgraph",
}
if revision is not None:
load_kwargs["revision"] = revision
model = AutoModelForCausalLM.from_pretrained(model_id, **load_kwargs)
model.eval()

from executorch.backends.mlx.llm.quantization import quantize_model_

quantize_model_(
model,
qlinear_config=qlinear,
qlinear_group_size=qlinear_group_size,
qembedding_config=qembedding,
qembedding_group_size=qembedding_group_size,
tie_word_embeddings=getattr(model.config, "tie_word_embeddings", False)
and not no_tie_word_embeddings,
)

exportable = OffGraphExportWrapper(model)

# The cache is built before the graph runs, so the runtime cannot learn its
# shape from the model. Publish it, one entry per cache:
# n_caches how many caches to allocate
# kv_heads KV heads per cache
# head_dims head dim per cache -- gemma-4 mixes 256 and 512
# windows sliding window per cache, 0 = flat
from executorch.backends.mlx.llm.cache import resolve_hf_cache_layout

# resolve_hf_cache_layout drops the KV-shared tail, so these are indexed by
# cache -- the same indexing _cache_id maps layers onto.
layer_types, cache_kv_heads, cache_head_dims = resolve_hf_cache_layout(model.config)
text_config = model.config.get_text_config()
sliding_window = getattr(text_config, "sliding_window", None) or 0
cache_windows = [
sliding_window if t == "sliding_attention" else 0 for t in layer_types
]
kv_metadata = {
# The cache count, not num_hidden_layers: gemma-4 E2B shares KV across
# its tail, so 15 caches for 35 layers.
"get_n_caches": len(layer_types),
"get_kv_heads": torch.tensor(cache_kv_heads, dtype=torch.int32),
"get_head_dims": torch.tensor(cache_head_dims, dtype=torch.int32),
"get_windows": torch.tensor(cache_windows, dtype=torch.int32),
}
logger.info(
f"KV cache layout: {len(layer_types)} caches, "
f"{sum(1 for w in cache_windows if w)} sliding (window {sliding_window})"
)

logger.info("Exporting model with torch.export...")
seq_length = 3
example_input_ids = torch.zeros((1, seq_length), dtype=torch.long)
example_cache_position = torch.arange(seq_length, dtype=torch.long)

seq_len_dim = torch.export.Dim("seq_length_dim", max=max_seq_len - 1)
dynamic_shapes = {
"input_ids": {1: seq_len_dim},
"cache_position": {0: seq_len_dim},
}

with torch.no_grad():
exported_program = torch.export.export(
exportable,
args=(),
kwargs={
"input_ids": example_input_ids,
"cache_position": example_cache_position,
},
dynamic_shapes=dynamic_shapes,
strict=True,
)

logger.info("Delegating to MLX backend...")
edge_program = exir.to_edge_transform_and_lower(
{"forward": exported_program},
transform_passes=get_default_passes(),
constant_methods=kv_metadata,
partitioner=[MLXPartitioner()],
compile_config=EdgeCompileConfig(
_check_ir_validity=False,
_skip_dim_order=True,
),
)

logger.info("Exporting to ExecuTorch...")
executorch_program = edge_program.to_executorch(
config=ExecutorchBackendConfig(
extract_delegate_segments=True,
memory_planning_pass=MemoryPlanningPass(alloc_graph_input=True),
)
)

_save_program(executorch_program, output_path)


def _save_program(executorch_program, output_path: str) -> None:
"""Save the ExecuTorch program to disk."""
os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True)
Expand All@@ -366,6 +508,7 @@ def export_llama_hf(
qembedding: Optional[str] = None,
use_custom_sdpa: bool = False,
use_custom_kv_cache: bool = False,
use_offgraph_cache: bool = False,
no_tie_word_embeddings: bool = False,
qlinear_group_size: Optional[int] = None,
qembedding_group_size: Optional[int] = None,
Expand All@@ -383,7 +526,26 @@ def export_llama_hf(
use_custom_sdpa: Use MLX custom SDPA (mlx::custom_sdpa)
use_custom_kv_cache: Use MLX custom KV cache (mlx::kv_cache_update)
"""
if use_custom_sdpa or use_custom_kv_cache:
if use_offgraph_cache:
if use_custom_sdpa or use_custom_kv_cache:
raise ValueError(
"--use-offgraph-cache is exclusive with --use-custom-sdpa / "
"--use-custom-kv-cache (it replaces both)"
)
logger.info("Using off-graph KV cache (update_and_attend)")
_export_with_offgraph_cache(
model_id=model_id,
revision=revision,
output_path=output_path,
max_seq_len=max_seq_len,
dtype=dtype,
qlinear=qlinear,
qembedding=qembedding,
no_tie_word_embeddings=no_tie_word_embeddings,
qlinear_group_size=qlinear_group_size,
qembedding_group_size=qembedding_group_size,
)
elif use_custom_sdpa or use_custom_kv_cache:
logger.info(
f"Using custom components: sdpa={use_custom_sdpa}, "
f"kv_cache={use_custom_kv_cache}"
Expand DownExpand Up@@ -468,6 +630,13 @@ def main():
default=False,
help="Use MLX custom KV cache (mlx::kv_cache_update)",
)
parser.add_argument(
"--use-offgraph-cache",
action="store_true",
default=False,
help="Use the off-graph KV cache op (kvcache::update_and_attend); "
"replaces --use-custom-sdpa/--use-custom-kv-cache",
)

args = parser.parse_args()

Expand All@@ -481,6 +650,7 @@ def main():
qembedding=args.qembedding,
use_custom_sdpa=args.use_custom_sdpa,
use_custom_kv_cache=args.use_custom_kv_cache,
use_offgraph_cache=args.use_offgraph_cache,
no_tie_word_embeddings=args.no_tie_word_embeddings,
qlinear_group_size=args.qlinear_group_size,
qembedding_group_size=args.qembedding_group_size,
Expand Down
4 changes: 3 additions & 1 deletion backends/mlx/llm/cache.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -525,7 +525,9 @@ def update(
# Current HF ExecuTorch wrappers copy the requested cache position
# into each StaticCache layer's cumulative_length before forward().
if hasattr(self.layers[layer_idx], "cumulative_length"):
cache_position = self.layers[layer_idx].cumulative_length
# cumulative_length is a scalar; KVCache.update indexes [0], so
# give it the 1-D shape a cache_kwargs caller would have passed.
cache_position = self.layers[layer_idx].cumulative_length.reshape(1)
else:
raise RuntimeError(
"cache_position was not provided and the pinned "
Expand Down
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
172 changes: 171 additions & 1 deletion backends/mlx/examples/llm/export_llm_hf.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -346,6 +346,148 @@ def _export_with_custom_components(
_save_program(executorch_program, output_path)


def _export_with_offgraph_cache(
model_id: str,
revision: Optional[str],
output_path: str,
max_seq_len: int,
dtype: str,
qlinear: Optional[str],
qembedding: Optional[str],
no_tie_word_embeddings: bool = False,
qlinear_group_size: Optional[int] = None,
qembedding_group_size: Optional[int] = None,
) -> None:
"""
Export using the off-graph KV cache op (kvcache::update_and_attend).

Unlike the custom-components path, the cache is not a graph tensor: the model
is run with use_cache=False (no StaticCache wrapper, no HFStaticCache swap),
so each attention layer emits an update_and_attend node fed this step's k/v.
The cache is created and bound at runtime via a cache_key.
"""
import executorch.exir as exir
from executorch.backends.mlx import MLXPartitioner
from executorch.backends.mlx.llm.hf_attention import (
OffGraphExportWrapper,
register_mlx_offgraph_attention,
)
from executorch.backends.mlx.passes import get_default_passes
from executorch.exir import EdgeCompileConfig
from executorch.exir.capture._config import ExecutorchBackendConfig
from executorch.exir.passes import MemoryPlanningPass
from transformers import AutoModelForCausalLM

torch_dtype_map = {
"fp32": torch.float32,
"fp16": torch.float16,
"bf16": torch.bfloat16,
}
torch_dtype = torch_dtype_map.get(dtype, torch.bfloat16)

register_mlx_offgraph_attention()
logger.info("Registered MLX off-graph attention (update_and_attend)")

logger.info(f"Loading HuggingFace model: {model_id}")
load_kwargs = {
"torch_dtype": torch_dtype,
"low_cpu_mem_usage": True,
"attn_implementation": "mlx_offgraph",
}
if revision is not None:
load_kwargs["revision"] = revision
model = AutoModelForCausalLM.from_pretrained(model_id, **load_kwargs)
model.eval()

from executorch.backends.mlx.llm.quantization import quantize_model_

quantize_model_(
model,
qlinear_config=qlinear,
qlinear_group_size=qlinear_group_size,
qembedding_config=qembedding,
qembedding_group_size=qembedding_group_size,
tie_word_embeddings=getattr(model.config, "tie_word_embeddings", False)
and not no_tie_word_embeddings,
)

exportable = OffGraphExportWrapper(model)

# The cache is built before the graph runs, so the runtime cannot learn its
# shape from the model. Publish it, one entry per cache:
# n_caches how many caches to allocate
# kv_heads KV heads per cache
# head_dims head dim per cache -- gemma-4 mixes 256 and 512
# windows sliding window per cache, 0 = flat
from executorch.backends.mlx.llm.cache import resolve_hf_cache_layout

# resolve_hf_cache_layout drops the KV-shared tail, so these are indexed by
# cache -- the same indexing _cache_id maps layers onto.
layer_types, cache_kv_heads, cache_head_dims = resolve_hf_cache_layout(model.config)
text_config = model.config.get_text_config()
sliding_window = getattr(text_config, "sliding_window", None) or 0
cache_windows = [
sliding_window if t == "sliding_attention" else 0 for t in layer_types
]
kv_metadata = {
# The cache count, not num_hidden_layers: gemma-4 E2B shares KV across
# its tail, so 15 caches for 35 layers.
"get_n_caches": len(layer_types),
"get_kv_heads": torch.tensor(cache_kv_heads, dtype=torch.int32),
"get_head_dims": torch.tensor(cache_head_dims, dtype=torch.int32),
"get_windows": torch.tensor(cache_windows, dtype=torch.int32),
}
logger.info(
f"KV cache layout: {len(layer_types)} caches, "
f"{sum(1 for w in cache_windows if w)} sliding (window {sliding_window})"
)

logger.info("Exporting model with torch.export...")
seq_length = 3
example_input_ids = torch.zeros((1, seq_length), dtype=torch.long)
example_cache_position = torch.arange(seq_length, dtype=torch.long)

seq_len_dim = torch.export.Dim("seq_length_dim", max=max_seq_len - 1)
dynamic_shapes = {
"input_ids": {1: seq_len_dim},
"cache_position": {0: seq_len_dim},
}

with torch.no_grad():
exported_program = torch.export.export(
exportable,
args=(),
kwargs={
"input_ids": example_input_ids,
"cache_position": example_cache_position,
},
dynamic_shapes=dynamic_shapes,
strict=True,
)

logger.info("Delegating to MLX backend...")
edge_program = exir.to_edge_transform_and_lower(
{"forward": exported_program},
transform_passes=get_default_passes(),
constant_methods=kv_metadata,
partitioner=[MLXPartitioner()],
compile_config=EdgeCompileConfig(
_check_ir_validity=False,
_skip_dim_order=True,
),
)

logger.info("Exporting to ExecuTorch...")
executorch_program = edge_program.to_executorch(
config=ExecutorchBackendConfig(
extract_delegate_segments=True,
memory_planning_pass=MemoryPlanningPass(alloc_graph_input=True),
)
)

_save_program(executorch_program, output_path)


def _save_program(executorch_program, output_path: str) -> None:
"""Save the ExecuTorch program to disk."""
os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True)
Expand All@@ -366,6 +508,7 @@ def export_llama_hf(
qembedding: Optional[str] = None,
use_custom_sdpa: bool = False,
use_custom_kv_cache: bool = False,
use_offgraph_cache: bool = False,
no_tie_word_embeddings: bool = False,
qlinear_group_size: Optional[int] = None,
qembedding_group_size: Optional[int] = None,
Expand All@@ -383,7 +526,26 @@ def export_llama_hf(
use_custom_sdpa: Use MLX custom SDPA (mlx::custom_sdpa)
use_custom_kv_cache: Use MLX custom KV cache (mlx::kv_cache_update)
"""
if use_custom_sdpa or use_custom_kv_cache:
if use_offgraph_cache:
if use_custom_sdpa or use_custom_kv_cache:
raise ValueError(
"--use-offgraph-cache is exclusive with --use-custom-sdpa / "
"--use-custom-kv-cache (it replaces both)"
)
logger.info("Using off-graph KV cache (update_and_attend)")
_export_with_offgraph_cache(
model_id=model_id,
revision=revision,
output_path=output_path,
max_seq_len=max_seq_len,
dtype=dtype,
qlinear=qlinear,
qembedding=qembedding,
no_tie_word_embeddings=no_tie_word_embeddings,
qlinear_group_size=qlinear_group_size,
qembedding_group_size=qembedding_group_size,
)
elif use_custom_sdpa or use_custom_kv_cache:
logger.info(
f"Using custom components: sdpa={use_custom_sdpa}, "
f"kv_cache={use_custom_kv_cache}"
Expand DownExpand Up@@ -468,6 +630,13 @@ def main():
default=False,
help="Use MLX custom KV cache (mlx::kv_cache_update)",
)
parser.add_argument(
"--use-offgraph-cache",
action="store_true",
default=False,
help="Use the off-graph KV cache op (kvcache::update_and_attend); "
"replaces --use-custom-sdpa/--use-custom-kv-cache",
)

args = parser.parse_args()

Expand All@@ -481,6 +650,7 @@ def main():
qembedding=args.qembedding,
use_custom_sdpa=args.use_custom_sdpa,
use_custom_kv_cache=args.use_custom_kv_cache,
use_offgraph_cache=args.use_offgraph_cache,
no_tie_word_embeddings=args.no_tie_word_embeddings,
qlinear_group_size=args.qlinear_group_size,
qembedding_group_size=args.qembedding_group_size,
Expand Down
4 changes: 3 additions & 1 deletion backends/mlx/llm/cache.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -525,7 +525,9 @@ def update(
# Current HF ExecuTorch wrappers copy the requested cache position
# into each StaticCache layer's cumulative_length before forward().
if hasattr(self.layers[layer_idx], "cumulative_length"):
cache_position = self.layers[layer_idx].cumulative_length
# cumulative_length is a scalar; KVCache.update indexes [0], so
# give it the 1-D shape a cache_kwargs caller would have passed.
cache_position = self.layers[layer_idx].cumulative_length.reshape(1)
else:
raise RuntimeError(
"cache_position was not provided and the pinned "
Expand Down
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
172 changes: 171 additions & 1 deletion backends/mlx/examples/llm/export_llm_hf.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -346,6 +346,148 @@ def _export_with_custom_components(
_save_program(executorch_program, output_path)


def _export_with_offgraph_cache(
model_id: str,
revision: Optional[str],
output_path: str,
max_seq_len: int,
dtype: str,
qlinear: Optional[str],
qembedding: Optional[str],
no_tie_word_embeddings: bool = False,
qlinear_group_size: Optional[int] = None,
qembedding_group_size: Optional[int] = None,
) -> None:
"""
Export using the off-graph KV cache op (kvcache::update_and_attend).

Unlike the custom-components path, the cache is not a graph tensor: the model
is run with use_cache=False (no StaticCache wrapper, no HFStaticCache swap),
so each attention layer emits an update_and_attend node fed this step's k/v.
The cache is created and bound at runtime via a cache_key.
"""
import executorch.exir as exir
from executorch.backends.mlx import MLXPartitioner
from executorch.backends.mlx.llm.hf_attention import (
OffGraphExportWrapper,
register_mlx_offgraph_attention,
)
from executorch.backends.mlx.passes import get_default_passes
from executorch.exir import EdgeCompileConfig
from executorch.exir.capture._config import ExecutorchBackendConfig
from executorch.exir.passes import MemoryPlanningPass
from transformers import AutoModelForCausalLM

torch_dtype_map = {
"fp32": torch.float32,
"fp16": torch.float16,
"bf16": torch.bfloat16,
}
torch_dtype = torch_dtype_map.get(dtype, torch.bfloat16)

register_mlx_offgraph_attention()
logger.info("Registered MLX off-graph attention (update_and_attend)")

logger.info(f"Loading HuggingFace model: {model_id}")
load_kwargs = {
"torch_dtype": torch_dtype,
"low_cpu_mem_usage": True,
"attn_implementation": "mlx_offgraph",
}
if revision is not None:
load_kwargs["revision"] = revision
model = AutoModelForCausalLM.from_pretrained(model_id, **load_kwargs)
model.eval()

from executorch.backends.mlx.llm.quantization import quantize_model_

quantize_model_(
model,
qlinear_config=qlinear,
qlinear_group_size=qlinear_group_size,
qembedding_config=qembedding,
qembedding_group_size=qembedding_group_size,
tie_word_embeddings=getattr(model.config, "tie_word_embeddings", False)
and not no_tie_word_embeddings,
)

exportable = OffGraphExportWrapper(model)

# The cache is built before the graph runs, so the runtime cannot learn its
# shape from the model. Publish it, one entry per cache:
# n_caches how many caches to allocate
# kv_heads KV heads per cache
# head_dims head dim per cache -- gemma-4 mixes 256 and 512
# windows sliding window per cache, 0 = flat
from executorch.backends.mlx.llm.cache import resolve_hf_cache_layout

# resolve_hf_cache_layout drops the KV-shared tail, so these are indexed by
# cache -- the same indexing _cache_id maps layers onto.
layer_types, cache_kv_heads, cache_head_dims = resolve_hf_cache_layout(model.config)
text_config = model.config.get_text_config()
sliding_window = getattr(text_config, "sliding_window", None) or 0
cache_windows = [
sliding_window if t == "sliding_attention" else 0 for t in layer_types
]
kv_metadata = {
# The cache count, not num_hidden_layers: gemma-4 E2B shares KV across
# its tail, so 15 caches for 35 layers.
"get_n_caches": len(layer_types),
"get_kv_heads": torch.tensor(cache_kv_heads, dtype=torch.int32),
"get_head_dims": torch.tensor(cache_head_dims, dtype=torch.int32),
"get_windows": torch.tensor(cache_windows, dtype=torch.int32),
}
logger.info(
f"KV cache layout: {len(layer_types)} caches, "
f"{sum(1 for w in cache_windows if w)} sliding (window {sliding_window})"
)

logger.info("Exporting model with torch.export...")
seq_length = 3
example_input_ids = torch.zeros((1, seq_length), dtype=torch.long)
example_cache_position = torch.arange(seq_length, dtype=torch.long)

seq_len_dim = torch.export.Dim("seq_length_dim", max=max_seq_len - 1)
dynamic_shapes = {
"input_ids": {1: seq_len_dim},
"cache_position": {0: seq_len_dim},
}

with torch.no_grad():
exported_program = torch.export.export(
exportable,
args=(),
kwargs={
"input_ids": example_input_ids,
"cache_position": example_cache_position,
},
dynamic_shapes=dynamic_shapes,
strict=True,
)

logger.info("Delegating to MLX backend...")
edge_program = exir.to_edge_transform_and_lower(
{"forward": exported_program},
transform_passes=get_default_passes(),
constant_methods=kv_metadata,
partitioner=[MLXPartitioner()],
compile_config=EdgeCompileConfig(
_check_ir_validity=False,
_skip_dim_order=True,
),
)

logger.info("Exporting to ExecuTorch...")
executorch_program = edge_program.to_executorch(
config=ExecutorchBackendConfig(
extract_delegate_segments=True,
memory_planning_pass=MemoryPlanningPass(alloc_graph_input=True),
)
)

_save_program(executorch_program, output_path)


def _save_program(executorch_program, output_path: str) -> None:
"""Save the ExecuTorch program to disk."""
os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True)
Expand All@@ -366,6 +508,7 @@ def export_llama_hf(
qembedding: Optional[str] = None,
use_custom_sdpa: bool = False,
use_custom_kv_cache: bool = False,
use_offgraph_cache: bool = False,
no_tie_word_embeddings: bool = False,
qlinear_group_size: Optional[int] = None,
qembedding_group_size: Optional[int] = None,
Expand All@@ -383,7 +526,26 @@ def export_llama_hf(
use_custom_sdpa: Use MLX custom SDPA (mlx::custom_sdpa)
use_custom_kv_cache: Use MLX custom KV cache (mlx::kv_cache_update)
"""
if use_custom_sdpa or use_custom_kv_cache:
if use_offgraph_cache:
if use_custom_sdpa or use_custom_kv_cache:
raise ValueError(
"--use-offgraph-cache is exclusive with --use-custom-sdpa / "
"--use-custom-kv-cache (it replaces both)"
)
logger.info("Using off-graph KV cache (update_and_attend)")
_export_with_offgraph_cache(
model_id=model_id,
revision=revision,
output_path=output_path,
max_seq_len=max_seq_len,
dtype=dtype,
qlinear=qlinear,
qembedding=qembedding,
no_tie_word_embeddings=no_tie_word_embeddings,
qlinear_group_size=qlinear_group_size,
qembedding_group_size=qembedding_group_size,
)
elif use_custom_sdpa or use_custom_kv_cache:
logger.info(
f"Using custom components: sdpa={use_custom_sdpa}, "
f"kv_cache={use_custom_kv_cache}"
Expand DownExpand Up@@ -468,6 +630,13 @@ def main():
default=False,
help="Use MLX custom KV cache (mlx::kv_cache_update)",
)
parser.add_argument(
"--use-offgraph-cache",
action="store_true",
default=False,
help="Use the off-graph KV cache op (kvcache::update_and_attend); "
"replaces --use-custom-sdpa/--use-custom-kv-cache",
)

args = parser.parse_args()

Expand All@@ -481,6 +650,7 @@ def main():
qembedding=args.qembedding,
use_custom_sdpa=args.use_custom_sdpa,
use_custom_kv_cache=args.use_custom_kv_cache,
use_offgraph_cache=args.use_offgraph_cache,
no_tie_word_embeddings=args.no_tie_word_embeddings,
qlinear_group_size=args.qlinear_group_size,
qembedding_group_size=args.qembedding_group_size,
Expand Down
4 changes: 3 additions & 1 deletion backends/mlx/llm/cache.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -525,7 +525,9 @@ def update(
# Current HF ExecuTorch wrappers copy the requested cache position
# into each StaticCache layer's cumulative_length before forward().
if hasattr(self.layers[layer_idx], "cumulative_length"):
cache_position = self.layers[layer_idx].cumulative_length
# cumulative_length is a scalar; KVCache.update indexes [0], so
# give it the 1-D shape a cache_kwargs caller would have passed.
cache_position = self.layers[layer_idx].cumulative_length.reshape(1)
else:
raise RuntimeError(
"cache_position was not provided and the pinned "
Expand Down
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
172 changes: 171 additions & 1 deletion backends/mlx/examples/llm/export_llm_hf.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -346,6 +346,148 @@ def _export_with_custom_components(
_save_program(executorch_program, output_path)


def _export_with_offgraph_cache(
model_id: str,
revision: Optional[str],
output_path: str,
max_seq_len: int,
dtype: str,
qlinear: Optional[str],
qembedding: Optional[str],
no_tie_word_embeddings: bool = False,
qlinear_group_size: Optional[int] = None,
qembedding_group_size: Optional[int] = None,
) -> None:
"""
Export using the off-graph KV cache op (kvcache::update_and_attend).

Unlike the custom-components path, the cache is not a graph tensor: the model
is run with use_cache=False (no StaticCache wrapper, no HFStaticCache swap),
so each attention layer emits an update_and_attend node fed this step's k/v.
The cache is created and bound at runtime via a cache_key.
"""
import executorch.exir as exir
from executorch.backends.mlx import MLXPartitioner
from executorch.backends.mlx.llm.hf_attention import (
OffGraphExportWrapper,
register_mlx_offgraph_attention,
)
from executorch.backends.mlx.passes import get_default_passes
from executorch.exir import EdgeCompileConfig
from executorch.exir.capture._config import ExecutorchBackendConfig
from executorch.exir.passes import MemoryPlanningPass
from transformers import AutoModelForCausalLM

torch_dtype_map = {
"fp32": torch.float32,
"fp16": torch.float16,
"bf16": torch.bfloat16,
}
torch_dtype = torch_dtype_map.get(dtype, torch.bfloat16)

register_mlx_offgraph_attention()
logger.info("Registered MLX off-graph attention (update_and_attend)")

logger.info(f"Loading HuggingFace model: {model_id}")
load_kwargs = {
"torch_dtype": torch_dtype,
"low_cpu_mem_usage": True,
"attn_implementation": "mlx_offgraph",
}
if revision is not None:
load_kwargs["revision"] = revision
model = AutoModelForCausalLM.from_pretrained(model_id, **load_kwargs)
model.eval()

from executorch.backends.mlx.llm.quantization import quantize_model_

quantize_model_(
model,
qlinear_config=qlinear,
qlinear_group_size=qlinear_group_size,
qembedding_config=qembedding,
qembedding_group_size=qembedding_group_size,
tie_word_embeddings=getattr(model.config, "tie_word_embeddings", False)
and not no_tie_word_embeddings,
)

exportable = OffGraphExportWrapper(model)

# The cache is built before the graph runs, so the runtime cannot learn its
# shape from the model. Publish it, one entry per cache:
# n_caches how many caches to allocate
# kv_heads KV heads per cache
# head_dims head dim per cache -- gemma-4 mixes 256 and 512
# windows sliding window per cache, 0 = flat
from executorch.backends.mlx.llm.cache import resolve_hf_cache_layout

# resolve_hf_cache_layout drops the KV-shared tail, so these are indexed by
# cache -- the same indexing _cache_id maps layers onto.
layer_types, cache_kv_heads, cache_head_dims = resolve_hf_cache_layout(model.config)
text_config = model.config.get_text_config()
sliding_window = getattr(text_config, "sliding_window", None) or 0
cache_windows = [
sliding_window if t == "sliding_attention" else 0 for t in layer_types
]
kv_metadata = {
# The cache count, not num_hidden_layers: gemma-4 E2B shares KV across
# its tail, so 15 caches for 35 layers.
"get_n_caches": len(layer_types),
"get_kv_heads": torch.tensor(cache_kv_heads, dtype=torch.int32),
"get_head_dims": torch.tensor(cache_head_dims, dtype=torch.int32),
"get_windows": torch.tensor(cache_windows, dtype=torch.int32),
}
logger.info(
f"KV cache layout: {len(layer_types)} caches, "
f"{sum(1 for w in cache_windows if w)} sliding (window {sliding_window})"
)

logger.info("Exporting model with torch.export...")
seq_length = 3
example_input_ids = torch.zeros((1, seq_length), dtype=torch.long)
example_cache_position = torch.arange(seq_length, dtype=torch.long)

seq_len_dim = torch.export.Dim("seq_length_dim", max=max_seq_len - 1)
dynamic_shapes = {
"input_ids": {1: seq_len_dim},
"cache_position": {0: seq_len_dim},
}

with torch.no_grad():
exported_program = torch.export.export(
exportable,
args=(),
kwargs={
"input_ids": example_input_ids,
"cache_position": example_cache_position,
},
dynamic_shapes=dynamic_shapes,
strict=True,
)

logger.info("Delegating to MLX backend...")
edge_program = exir.to_edge_transform_and_lower(
{"forward": exported_program},
transform_passes=get_default_passes(),
constant_methods=kv_metadata,
partitioner=[MLXPartitioner()],
compile_config=EdgeCompileConfig(
_check_ir_validity=False,
_skip_dim_order=True,
),
)

logger.info("Exporting to ExecuTorch...")
executorch_program = edge_program.to_executorch(
config=ExecutorchBackendConfig(
extract_delegate_segments=True,
memory_planning_pass=MemoryPlanningPass(alloc_graph_input=True),
)
)

_save_program(executorch_program, output_path)


def _save_program(executorch_program, output_path: str) -> None:
"""Save the ExecuTorch program to disk."""
os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True)
Expand All@@ -366,6 +508,7 @@ def export_llama_hf(
qembedding: Optional[str] = None,
use_custom_sdpa: bool = False,
use_custom_kv_cache: bool = False,
use_offgraph_cache: bool = False,
no_tie_word_embeddings: bool = False,
qlinear_group_size: Optional[int] = None,
qembedding_group_size: Optional[int] = None,
Expand All@@ -383,7 +526,26 @@ def export_llama_hf(
use_custom_sdpa: Use MLX custom SDPA (mlx::custom_sdpa)
use_custom_kv_cache: Use MLX custom KV cache (mlx::kv_cache_update)
"""
if use_custom_sdpa or use_custom_kv_cache:
if use_offgraph_cache:
if use_custom_sdpa or use_custom_kv_cache:
raise ValueError(
"--use-offgraph-cache is exclusive with --use-custom-sdpa / "
"--use-custom-kv-cache (it replaces both)"
)
logger.info("Using off-graph KV cache (update_and_attend)")
_export_with_offgraph_cache(
model_id=model_id,
revision=revision,
output_path=output_path,
max_seq_len=max_seq_len,
dtype=dtype,
qlinear=qlinear,
qembedding=qembedding,
no_tie_word_embeddings=no_tie_word_embeddings,
qlinear_group_size=qlinear_group_size,
qembedding_group_size=qembedding_group_size,
)
elif use_custom_sdpa or use_custom_kv_cache:
logger.info(
f"Using custom components: sdpa={use_custom_sdpa}, "
f"kv_cache={use_custom_kv_cache}"
Expand DownExpand Up@@ -468,6 +630,13 @@ def main():
default=False,
help="Use MLX custom KV cache (mlx::kv_cache_update)",
)
parser.add_argument(
"--use-offgraph-cache",
action="store_true",
default=False,
help="Use the off-graph KV cache op (kvcache::update_and_attend); "
"replaces --use-custom-sdpa/--use-custom-kv-cache",
)

args = parser.parse_args()

Expand All@@ -481,6 +650,7 @@ def main():
qembedding=args.qembedding,
use_custom_sdpa=args.use_custom_sdpa,
use_custom_kv_cache=args.use_custom_kv_cache,
use_offgraph_cache=args.use_offgraph_cache,
no_tie_word_embeddings=args.no_tie_word_embeddings,
qlinear_group_size=args.qlinear_group_size,
qembedding_group_size=args.qembedding_group_size,
Expand Down
4 changes: 3 additions & 1 deletion backends/mlx/llm/cache.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -525,7 +525,9 @@ def update(
# Current HF ExecuTorch wrappers copy the requested cache position
# into each StaticCache layer's cumulative_length before forward().
if hasattr(self.layers[layer_idx], "cumulative_length"):
cache_position = self.layers[layer_idx].cumulative_length
# cumulative_length is a scalar; KVCache.update indexes [0], so
# give it the 1-D shape a cache_kwargs caller would have passed.
cache_position = self.layers[layer_idx].cumulative_length.reshape(1)
else:
raise RuntimeError(
"cache_position was not provided and the pinned "
Expand Down
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
172 changes: 171 additions & 1 deletion backends/mlx/examples/llm/export_llm_hf.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -346,6 +346,148 @@ def _export_with_custom_components(
_save_program(executorch_program, output_path)


def _export_with_offgraph_cache(
model_id: str,
revision: Optional[str],
output_path: str,
max_seq_len: int,
dtype: str,
qlinear: Optional[str],
qembedding: Optional[str],
no_tie_word_embeddings: bool = False,
qlinear_group_size: Optional[int] = None,
qembedding_group_size: Optional[int] = None,
) -> None:
"""
Export using the off-graph KV cache op (kvcache::update_and_attend).

Unlike the custom-components path, the cache is not a graph tensor: the model
is run with use_cache=False (no StaticCache wrapper, no HFStaticCache swap),
so each attention layer emits an update_and_attend node fed this step's k/v.
The cache is created and bound at runtime via a cache_key.
"""
import executorch.exir as exir
from executorch.backends.mlx import MLXPartitioner
from executorch.backends.mlx.llm.hf_attention import (
OffGraphExportWrapper,
register_mlx_offgraph_attention,
)
from executorch.backends.mlx.passes import get_default_passes
from executorch.exir import EdgeCompileConfig
from executorch.exir.capture._config import ExecutorchBackendConfig
from executorch.exir.passes import MemoryPlanningPass
from transformers import AutoModelForCausalLM

torch_dtype_map = {
"fp32": torch.float32,
"fp16": torch.float16,
"bf16": torch.bfloat16,
}
torch_dtype = torch_dtype_map.get(dtype, torch.bfloat16)

register_mlx_offgraph_attention()
logger.info("Registered MLX off-graph attention (update_and_attend)")

logger.info(f"Loading HuggingFace model: {model_id}")
load_kwargs = {
"torch_dtype": torch_dtype,
"low_cpu_mem_usage": True,
"attn_implementation": "mlx_offgraph",
}
if revision is not None:
load_kwargs["revision"] = revision
model = AutoModelForCausalLM.from_pretrained(model_id, **load_kwargs)
model.eval()

from executorch.backends.mlx.llm.quantization import quantize_model_

quantize_model_(
model,
qlinear_config=qlinear,
qlinear_group_size=qlinear_group_size,
qembedding_config=qembedding,
qembedding_group_size=qembedding_group_size,
tie_word_embeddings=getattr(model.config, "tie_word_embeddings", False)
and not no_tie_word_embeddings,
)

exportable = OffGraphExportWrapper(model)

# The cache is built before the graph runs, so the runtime cannot learn its
# shape from the model. Publish it, one entry per cache:
# n_caches how many caches to allocate
# kv_heads KV heads per cache
# head_dims head dim per cache -- gemma-4 mixes 256 and 512
# windows sliding window per cache, 0 = flat
from executorch.backends.mlx.llm.cache import resolve_hf_cache_layout

# resolve_hf_cache_layout drops the KV-shared tail, so these are indexed by
# cache -- the same indexing _cache_id maps layers onto.
layer_types, cache_kv_heads, cache_head_dims = resolve_hf_cache_layout(model.config)
text_config = model.config.get_text_config()
sliding_window = getattr(text_config, "sliding_window", None) or 0
cache_windows = [
sliding_window if t == "sliding_attention" else 0 for t in layer_types
]
kv_metadata = {
# The cache count, not num_hidden_layers: gemma-4 E2B shares KV across
# its tail, so 15 caches for 35 layers.
"get_n_caches": len(layer_types),
"get_kv_heads": torch.tensor(cache_kv_heads, dtype=torch.int32),
"get_head_dims": torch.tensor(cache_head_dims, dtype=torch.int32),
"get_windows": torch.tensor(cache_windows, dtype=torch.int32),
}
logger.info(
f"KV cache layout: {len(layer_types)} caches, "
f"{sum(1 for w in cache_windows if w)} sliding (window {sliding_window})"
)

logger.info("Exporting model with torch.export...")
seq_length = 3
example_input_ids = torch.zeros((1, seq_length), dtype=torch.long)
example_cache_position = torch.arange(seq_length, dtype=torch.long)

seq_len_dim = torch.export.Dim("seq_length_dim", max=max_seq_len - 1)
dynamic_shapes = {
"input_ids": {1: seq_len_dim},
"cache_position": {0: seq_len_dim},
}

with torch.no_grad():
exported_program = torch.export.export(
exportable,
args=(),
kwargs={
"input_ids": example_input_ids,
"cache_position": example_cache_position,
},
dynamic_shapes=dynamic_shapes,
strict=True,
)

logger.info("Delegating to MLX backend...")
edge_program = exir.to_edge_transform_and_lower(
{"forward": exported_program},
transform_passes=get_default_passes(),
constant_methods=kv_metadata,
partitioner=[MLXPartitioner()],
compile_config=EdgeCompileConfig(
_check_ir_validity=False,
_skip_dim_order=True,
),
)

logger.info("Exporting to ExecuTorch...")
executorch_program = edge_program.to_executorch(
config=ExecutorchBackendConfig(
extract_delegate_segments=True,
memory_planning_pass=MemoryPlanningPass(alloc_graph_input=True),
)
)

_save_program(executorch_program, output_path)


def _save_program(executorch_program, output_path: str) -> None:
"""Save the ExecuTorch program to disk."""
os.makedirs(os.path.dirname(output_path) or ".", exist_ok=True)
Expand All@@ -366,6 +508,7 @@ def export_llama_hf(
qembedding: Optional[str] = None,
use_custom_sdpa: bool = False,
use_custom_kv_cache: bool = False,
use_offgraph_cache: bool = False,
no_tie_word_embeddings: bool = False,
qlinear_group_size: Optional[int] = None,
qembedding_group_size: Optional[int] = None,
Expand All@@ -383,7 +526,26 @@ def export_llama_hf(
use_custom_sdpa: Use MLX custom SDPA (mlx::custom_sdpa)
use_custom_kv_cache: Use MLX custom KV cache (mlx::kv_cache_update)
"""
if use_custom_sdpa or use_custom_kv_cache:
if use_offgraph_cache:
if use_custom_sdpa or use_custom_kv_cache:
raise ValueError(
"--use-offgraph-cache is exclusive with --use-custom-sdpa / "
"--use-custom-kv-cache (it replaces both)"
)
logger.info("Using off-graph KV cache (update_and_attend)")
_export_with_offgraph_cache(
model_id=model_id,
revision=revision,
output_path=output_path,
max_seq_len=max_seq_len,
dtype=dtype,
qlinear=qlinear,
qembedding=qembedding,
no_tie_word_embeddings=no_tie_word_embeddings,
qlinear_group_size=qlinear_group_size,
qembedding_group_size=qembedding_group_size,
)
elif use_custom_sdpa or use_custom_kv_cache:
logger.info(
f"Using custom components: sdpa={use_custom_sdpa}, "
f"kv_cache={use_custom_kv_cache}"
Expand DownExpand Up@@ -468,6 +630,13 @@ def main():
default=False,
help="Use MLX custom KV cache (mlx::kv_cache_update)",
)
parser.add_argument(
"--use-offgraph-cache",
action="store_true",
default=False,
help="Use the off-graph KV cache op (kvcache::update_and_attend); "
"replaces --use-custom-sdpa/--use-custom-kv-cache",
)

args = parser.parse_args()

Expand All@@ -481,6 +650,7 @@ def main():
qembedding=args.qembedding,
use_custom_sdpa=args.use_custom_sdpa,
use_custom_kv_cache=args.use_custom_kv_cache,
use_offgraph_cache=args.use_offgraph_cache,
no_tie_word_embeddings=args.no_tie_word_embeddings,
qlinear_group_size=args.qlinear_group_size,
qembedding_group_size=args.qembedding_group_size,
Expand Down
4 changes: 3 additions & 1 deletion backends/mlx/llm/cache.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -525,7 +525,9 @@ def update(
# Current HF ExecuTorch wrappers copy the requested cache position
# into each StaticCache layer's cumulative_length before forward().
if hasattr(self.layers[layer_idx], "cumulative_length"):
cache_position = self.layers[layer_idx].cumulative_length
# cumulative_length is a scalar; KVCache.update indexes [0], so
# give it the 1-D shape a cache_kwargs caller would have passed.
cache_position = self.layers[layer_idx].cumulative_length.reshape(1)
else:
raise RuntimeError(
"cache_position was not provided and the pinned "
Expand Down
Loading
Loading