Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion gpu_test/conftest.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -118,7 +118,10 @@ def _cleanup_orphans(self) -> None:

def _launch(self) -> None:
"""Find the cheapest suitable offer and launch an instance."""
query = f"num_gpus=1 rentable=True rented=False compute_cap>=700 dph<={MAX_COST_PER_HOUR}"
query = (
f"num_gpus=1 rentable=True rented=False compute_cap>=700"
f" reliability2>=0.95 inet_up>=100 dph<={MAX_COST_PER_HOUR}"
)
offers = self.sdk.search_offers(query=query, order="dph", limit=5)

if not offers:
Expand Down
155 changes: 155 additions & 0 deletions gpu_test/test_kernels.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@

from typing import TYPE_CHECKING

import numpy as np
import pytest

if TYPE_CHECKING:
Expand DownExpand Up@@ -583,3 +584,157 @@ def test_float_to_int_conversion(kernel_runner: KernelRunner) -> None:
forth_source=("\\! kernel main\n\\! param DATA i64[256]\n7.9 F>S\n0 CELLS DATA + !"),
)
assert result[0] == 7


# --- Attention ---

_ATTENTION_KERNEL = """\
\\! kernel attention
\\! param Q f64[{n}]
\\! param K f64[{n}]
\\! param V f64[{n}]
\\! param O f64[{n}]
\\! param SEQ_LEN i64
\\! param HEAD_DIM i64
\\! shared SCORES f64[{seq_len}]
\\! shared SCRATCH f64[{seq_len}]
BID-X
TID-X
0.0
HEAD_DIM 0 DO
2 PICK HEAD_DIM * I + CELLS Q + F@
2 PICK HEAD_DIM * I + CELLS K + F@
F* F+
LOOP
HEAD_DIM S>F FSQRT F/
OVER 3 PICK >
IF DROP -1.0e30 THEN
OVER CELLS SCORES + SF!
BARRIER
TID-X 0= IF
0 CELLS SCORES + SF@
SEQ_LEN 1 DO I CELLS SCORES + SF@ FMAX LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F- FEXP
OVER CELLS SCORES + SF!
BARRIER
TID-X 0= IF
0.0
SEQ_LEN 0 DO I CELLS SCORES + SF@ F+ LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F/
OVER CELLS SCORES + SF!
BARRIER
DUP BEGIN DUP HEAD_DIM < WHILE
0.0
SEQ_LEN 0 DO
I CELLS SCORES + SF@
I HEAD_DIM * 3 PICK + CELLS V + F@
F* F+
LOOP
OVER 4 PICK HEAD_DIM * + CELLS O + F!
BDIM-X +
REPEAT
DROP DROP DROP
"""


def _attention_reference(q: np.ndarray, k: np.ndarray, v: np.ndarray, seq_len: int) -> list[float]:
"""Compute scaled dot-product attention with causal mask (NumPy reference)."""
head_dim = q.shape[1]
scores = q @ k.T / np.sqrt(head_dim)
causal_mask = np.triu(np.ones((seq_len, seq_len), dtype=bool), k=1)
scores[causal_mask] = -1e30
exp_scores = np.exp(scores - scores.max(axis=1, keepdims=True))
attn = exp_scores / exp_scores.sum(axis=1, keepdims=True)
return (attn @ v).flatten().tolist()


def test_naive_attention_f64(kernel_runner: KernelRunner) -> None:
"""Naive scaled dot-product attention with causal mask.

O = softmax(Q @ K^T / sqrt(d_k)) @ V, seq_len=4, head_dim=4.
One block per query row, one thread per key position.
"""
seq_len, head_dim = 4, 4

q = np.array(
[
[1.0, 0.0, 1.0, 0.0],
[0.0, 1.0, 0.0, 1.0],
[1.0, 1.0, 0.0, 0.0],
[0.0, 0.0, 1.0, 1.0],
]
)
k = np.array(
[
[1.0, 0.0, 0.0, 1.0],
[0.0, 1.0, 1.0, 0.0],
[1.0, 1.0, 0.0, 0.0],
[0.0, 0.0, 1.0, 1.0],
]
)
v = np.array(
[
[1.0, 2.0, 3.0, 4.0],
[5.0, 6.0, 7.0, 8.0],
[9.0, 10.0, 11.0, 12.0],
[13.0, 14.0, 15.0, 16.0],
]
)

expected = _attention_reference(q, k, v, seq_len)
n = seq_len * head_dim

result = kernel_runner.run(
forth_source=_ATTENTION_KERNEL.format(n=n, seq_len=seq_len),
params={
"Q": q.flatten().tolist(),
"K": k.flatten().tolist(),
"V": v.flatten().tolist(),
"SEQ_LEN": seq_len,
"HEAD_DIM": head_dim,
},
grid=(seq_len, 1, 1),
block=(seq_len, 1, 1),
output_param=3,
output_count=n,
)
assert result == [pytest.approx(v) for v in expected]


def test_naive_attention_f64_16x64(kernel_runner: KernelRunner) -> None:
"""Naive scaled dot-product attention, seq_len=16, head_dim=64."""
seq_len, head_dim = 16, 64

rng = np.random.default_rng(42)
q = rng.standard_normal((seq_len, head_dim))
k = rng.standard_normal((seq_len, head_dim))
v = rng.standard_normal((seq_len, head_dim))

expected = _attention_reference(q, k, v, seq_len)
n = seq_len * head_dim

result = kernel_runner.run(
forth_source=_ATTENTION_KERNEL.format(n=n, seq_len=seq_len),
params={
"Q": q.flatten().tolist(),
"K": k.flatten().tolist(),
"V": v.flatten().tolist(),
"SEQ_LEN": seq_len,
"HEAD_DIM": head_dim,
},
grid=(seq_len, 1, 1),
block=(seq_len, 1, 1),
output_param=3,
output_count=n,
)
assert result == [pytest.approx(v) for v in expected]
5 changes: 5 additions & 0 deletions lib/Bitcode/CMakeLists.txt
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,5 @@
set(WARPFORTH_LIBDEVICE_PATH
"${CMAKE_CURRENT_BINARY_DIR}/libdevice.10.bc" PARENT_SCOPE)

configure_file(libdevice.10.bc
"${CMAKE_CURRENT_BINARY_DIR}/libdevice.10.bc" COPYONLY)
Binary file addedlib/Bitcode/libdevice.10.bc
Binary file not shown.
1 change: 1 addition & 0 deletions lib/CMakeLists.txt
Original file line numberDiff line numberDiff line change
@@ -1,3 +1,4 @@
add_subdirectory(Dialect)
add_subdirectory(Bitcode)
add_subdirectory(Conversion)
add_subdirectory(Translation)
4 changes: 4 additions & 0 deletions lib/Conversion/CMakeLists.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,5 +16,9 @@ add_mlir_library(MLIRConversionPasses
MLIRTransforms
)

target_compile_definitions(obj.MLIRConversionPasses PRIVATE
WARPFORTH_LIBDEVICE_PATH="${WARPFORTH_LIBDEVICE_PATH}"
)

add_subdirectory(ForthToMemRef)
add_subdirectory(ForthToGPU)
5 changes: 4 additions & 1 deletion lib/Conversion/Passes.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -33,7 +33,10 @@ void buildWarpForthPipeline(OpPassManager &pm) {
pm.addPass(createCanonicalizerPass());

// Stage 4: Attach NVVM target to GPU modules (sm_70 = Volta architecture)
pm.addPass(createGpuNVVMAttachTarget());
GpuNVVMAttachTargetOptions nvvmOptions;
nvvmOptions.chip = "sm_70";
nvvmOptions.linkLibs.push_back(WARPFORTH_LIBDEVICE_PATH);
pm.addPass(createGpuNVVMAttachTarget(nvvmOptions));

// Stage 5: Lower GPU to NVVM with bare pointers
ConvertGpuOpsToNVVMOpsOptions gpuToNVVMOptions;
Expand Down
1 change: 1 addition & 0 deletions pyproject.toml
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@ version = "0.1.0"
requires-python = ">=3.11"
dependencies = [
"lit>=18.1.0",
"numpy",
"pytest",
"vastai-sdk",
]
Expand Down
80 changes: 80 additions & 0 deletions test/Pipeline/attention.forth
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,80 @@
\ RUN: %warpforth-translate --forth-to-mlir %s | %warpforth-opt --warpforth-pipeline | %FileCheck %s

\ Verify that a naive attention kernel with shared memory and float intrinsics
\ survives the full pipeline to gpu.binary.
\ CHECK: gpu.binary @warpforth_module

\! kernel attention
\! param Q f64[16]
\! param K f64[16]
\! param V f64[16]
\! param O f64[16]
\! param SEQ_LEN i64
\! param HEAD_DIM i64
\! shared SCORES f64[4]
\! shared SCRATCH f64[4]

\ row = BID-X, t = TID-X
BID-X
TID-X

\ --- Dot product: Q[row,:] . K[t,:] ---
0.0
HEAD_DIM 0 DO
2 PICK HEAD_DIM * I + CELLS Q + F@
2 PICK HEAD_DIM * I + CELLS K + F@
F* F+
LOOP
HEAD_DIM S>F FSQRT F/

\ --- Causal mask: if t > row, score = -inf ---
OVER 3 PICK >
IF DROP -1.0e30 THEN

\ --- Store score to shared memory ---
OVER CELLS SCORES + SF!
BARRIER

\ --- Softmax: max reduction (thread 0) ---
TID-X 0= IF
0 CELLS SCORES + SF@
SEQ_LEN 1 DO I CELLS SCORES + SF@ FMAX LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER

\ --- Softmax: exp(score - max) ---
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F- FEXP
OVER CELLS SCORES + SF!
BARRIER

\ --- Softmax: sum reduction (thread 0) ---
TID-X 0= IF
0.0
SEQ_LEN 0 DO I CELLS SCORES + SF@ F+ LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER

\ --- Softmax: normalize ---
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F/
OVER CELLS SCORES + SF!
BARRIER

\ --- V accumulation: O[row,col] = sum_j SCORES[j] * V[j*HD + col] ---
\ Stride over head_dim columns: col = t, t+BDIM-X, t+2*BDIM-X, ...
DUP BEGIN DUP HEAD_DIM < WHILE
0.0
SEQ_LEN 0 DO
I CELLS SCORES + SF@
I HEAD_DIM * 3 PICK + CELLS V + F@
F* F+
LOOP
OVER 4 PICK HEAD_DIM * + CELLS O + F!
BDIM-X +
REPEAT
DROP DROP DROP
2 changes: 2 additions & 0 deletions uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion gpu_test/conftest.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -118,7 +118,10 @@ def _cleanup_orphans(self) -> None:

def _launch(self) -> None:
"""Find the cheapest suitable offer and launch an instance."""
query = f"num_gpus=1 rentable=True rented=False compute_cap>=700 dph<={MAX_COST_PER_HOUR}"
query = (
f"num_gpus=1 rentable=True rented=False compute_cap>=700"
f" reliability2>=0.95 inet_up>=100 dph<={MAX_COST_PER_HOUR}"
)
offers = self.sdk.search_offers(query=query, order="dph", limit=5)

if not offers:
Expand Down
155 changes: 155 additions & 0 deletions gpu_test/test_kernels.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@

from typing import TYPE_CHECKING

import numpy as np
import pytest

if TYPE_CHECKING:
Expand DownExpand Up@@ -583,3 +584,157 @@ def test_float_to_int_conversion(kernel_runner: KernelRunner) -> None:
forth_source=("\\! kernel main\n\\! param DATA i64[256]\n7.9 F>S\n0 CELLS DATA + !"),
)
assert result[0] == 7


# --- Attention ---

_ATTENTION_KERNEL = """\
\\! kernel attention
\\! param Q f64[{n}]
\\! param K f64[{n}]
\\! param V f64[{n}]
\\! param O f64[{n}]
\\! param SEQ_LEN i64
\\! param HEAD_DIM i64
\\! shared SCORES f64[{seq_len}]
\\! shared SCRATCH f64[{seq_len}]
BID-X
TID-X
0.0
HEAD_DIM 0 DO
2 PICK HEAD_DIM * I + CELLS Q + F@
2 PICK HEAD_DIM * I + CELLS K + F@
F* F+
LOOP
HEAD_DIM S>F FSQRT F/
OVER 3 PICK >
IF DROP -1.0e30 THEN
OVER CELLS SCORES + SF!
BARRIER
TID-X 0= IF
0 CELLS SCORES + SF@
SEQ_LEN 1 DO I CELLS SCORES + SF@ FMAX LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F- FEXP
OVER CELLS SCORES + SF!
BARRIER
TID-X 0= IF
0.0
SEQ_LEN 0 DO I CELLS SCORES + SF@ F+ LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F/
OVER CELLS SCORES + SF!
BARRIER
DUP BEGIN DUP HEAD_DIM < WHILE
0.0
SEQ_LEN 0 DO
I CELLS SCORES + SF@
I HEAD_DIM * 3 PICK + CELLS V + F@
F* F+
LOOP
OVER 4 PICK HEAD_DIM * + CELLS O + F!
BDIM-X +
REPEAT
DROP DROP DROP
"""


def _attention_reference(q: np.ndarray, k: np.ndarray, v: np.ndarray, seq_len: int) -> list[float]:
"""Compute scaled dot-product attention with causal mask (NumPy reference)."""
head_dim = q.shape[1]
scores = q @ k.T / np.sqrt(head_dim)
causal_mask = np.triu(np.ones((seq_len, seq_len), dtype=bool), k=1)
scores[causal_mask] = -1e30
exp_scores = np.exp(scores - scores.max(axis=1, keepdims=True))
attn = exp_scores / exp_scores.sum(axis=1, keepdims=True)
return (attn @ v).flatten().tolist()


def test_naive_attention_f64(kernel_runner: KernelRunner) -> None:
"""Naive scaled dot-product attention with causal mask.

O = softmax(Q @ K^T / sqrt(d_k)) @ V, seq_len=4, head_dim=4.
One block per query row, one thread per key position.
"""
seq_len, head_dim = 4, 4

q = np.array(
[
[1.0, 0.0, 1.0, 0.0],
[0.0, 1.0, 0.0, 1.0],
[1.0, 1.0, 0.0, 0.0],
[0.0, 0.0, 1.0, 1.0],
]
)
k = np.array(
[
[1.0, 0.0, 0.0, 1.0],
[0.0, 1.0, 1.0, 0.0],
[1.0, 1.0, 0.0, 0.0],
[0.0, 0.0, 1.0, 1.0],
]
)
v = np.array(
[
[1.0, 2.0, 3.0, 4.0],
[5.0, 6.0, 7.0, 8.0],
[9.0, 10.0, 11.0, 12.0],
[13.0, 14.0, 15.0, 16.0],
]
)

expected = _attention_reference(q, k, v, seq_len)
n = seq_len * head_dim

result = kernel_runner.run(
forth_source=_ATTENTION_KERNEL.format(n=n, seq_len=seq_len),
params={
"Q": q.flatten().tolist(),
"K": k.flatten().tolist(),
"V": v.flatten().tolist(),
"SEQ_LEN": seq_len,
"HEAD_DIM": head_dim,
},
grid=(seq_len, 1, 1),
block=(seq_len, 1, 1),
output_param=3,
output_count=n,
)
assert result == [pytest.approx(v) for v in expected]


def test_naive_attention_f64_16x64(kernel_runner: KernelRunner) -> None:
"""Naive scaled dot-product attention, seq_len=16, head_dim=64."""
seq_len, head_dim = 16, 64

rng = np.random.default_rng(42)
q = rng.standard_normal((seq_len, head_dim))
k = rng.standard_normal((seq_len, head_dim))
v = rng.standard_normal((seq_len, head_dim))

expected = _attention_reference(q, k, v, seq_len)
n = seq_len * head_dim

result = kernel_runner.run(
forth_source=_ATTENTION_KERNEL.format(n=n, seq_len=seq_len),
params={
"Q": q.flatten().tolist(),
"K": k.flatten().tolist(),
"V": v.flatten().tolist(),
"SEQ_LEN": seq_len,
"HEAD_DIM": head_dim,
},
grid=(seq_len, 1, 1),
block=(seq_len, 1, 1),
output_param=3,
output_count=n,
)
assert result == [pytest.approx(v) for v in expected]
5 changes: 5 additions & 0 deletions lib/Bitcode/CMakeLists.txt
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,5 @@
set(WARPFORTH_LIBDEVICE_PATH
"${CMAKE_CURRENT_BINARY_DIR}/libdevice.10.bc" PARENT_SCOPE)

configure_file(libdevice.10.bc
"${CMAKE_CURRENT_BINARY_DIR}/libdevice.10.bc" COPYONLY)
Binary file addedlib/Bitcode/libdevice.10.bc
Binary file not shown.
1 change: 1 addition & 0 deletions lib/CMakeLists.txt
Original file line numberDiff line numberDiff line change
@@ -1,3 +1,4 @@
add_subdirectory(Dialect)
add_subdirectory(Bitcode)
add_subdirectory(Conversion)
add_subdirectory(Translation)
4 changes: 4 additions & 0 deletions lib/Conversion/CMakeLists.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,5 +16,9 @@ add_mlir_library(MLIRConversionPasses
MLIRTransforms
)

target_compile_definitions(obj.MLIRConversionPasses PRIVATE
WARPFORTH_LIBDEVICE_PATH="${WARPFORTH_LIBDEVICE_PATH}"
)

add_subdirectory(ForthToMemRef)
add_subdirectory(ForthToGPU)
5 changes: 4 additions & 1 deletion lib/Conversion/Passes.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -33,7 +33,10 @@ void buildWarpForthPipeline(OpPassManager &pm) {
pm.addPass(createCanonicalizerPass());

// Stage 4: Attach NVVM target to GPU modules (sm_70 = Volta architecture)
pm.addPass(createGpuNVVMAttachTarget());
GpuNVVMAttachTargetOptions nvvmOptions;
nvvmOptions.chip = "sm_70";
nvvmOptions.linkLibs.push_back(WARPFORTH_LIBDEVICE_PATH);
pm.addPass(createGpuNVVMAttachTarget(nvvmOptions));

// Stage 5: Lower GPU to NVVM with bare pointers
ConvertGpuOpsToNVVMOpsOptions gpuToNVVMOptions;
Expand Down
1 change: 1 addition & 0 deletions pyproject.toml
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@ version = "0.1.0"
requires-python = ">=3.11"
dependencies = [
"lit>=18.1.0",
"numpy",
"pytest",
"vastai-sdk",
]
Expand Down
80 changes: 80 additions & 0 deletions test/Pipeline/attention.forth
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,80 @@
\ RUN: %warpforth-translate --forth-to-mlir %s | %warpforth-opt --warpforth-pipeline | %FileCheck %s

\ Verify that a naive attention kernel with shared memory and float intrinsics
\ survives the full pipeline to gpu.binary.
\ CHECK: gpu.binary @warpforth_module

\! kernel attention
\! param Q f64[16]
\! param K f64[16]
\! param V f64[16]
\! param O f64[16]
\! param SEQ_LEN i64
\! param HEAD_DIM i64
\! shared SCORES f64[4]
\! shared SCRATCH f64[4]

\ row = BID-X, t = TID-X
BID-X
TID-X

\ --- Dot product: Q[row,:] . K[t,:] ---
0.0
HEAD_DIM 0 DO
2 PICK HEAD_DIM * I + CELLS Q + F@
2 PICK HEAD_DIM * I + CELLS K + F@
F* F+
LOOP
HEAD_DIM S>F FSQRT F/

\ --- Causal mask: if t > row, score = -inf ---
OVER 3 PICK >
IF DROP -1.0e30 THEN

\ --- Store score to shared memory ---
OVER CELLS SCORES + SF!
BARRIER

\ --- Softmax: max reduction (thread 0) ---
TID-X 0= IF
0 CELLS SCORES + SF@
SEQ_LEN 1 DO I CELLS SCORES + SF@ FMAX LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER

\ --- Softmax: exp(score - max) ---
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F- FEXP
OVER CELLS SCORES + SF!
BARRIER

\ --- Softmax: sum reduction (thread 0) ---
TID-X 0= IF
0.0
SEQ_LEN 0 DO I CELLS SCORES + SF@ F+ LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER

\ --- Softmax: normalize ---
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F/
OVER CELLS SCORES + SF!
BARRIER

\ --- V accumulation: O[row,col] = sum_j SCORES[j] * V[j*HD + col] ---
\ Stride over head_dim columns: col = t, t+BDIM-X, t+2*BDIM-X, ...
DUP BEGIN DUP HEAD_DIM < WHILE
0.0
SEQ_LEN 0 DO
I CELLS SCORES + SF@
I HEAD_DIM * 3 PICK + CELLS V + F@
F* F+
LOOP
OVER 4 PICK HEAD_DIM * + CELLS O + F!
BDIM-X +
REPEAT
DROP DROP DROP
2 changes: 2 additions & 0 deletions uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion gpu_test/conftest.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -118,7 +118,10 @@ def _cleanup_orphans(self) -> None:

def _launch(self) -> None:
"""Find the cheapest suitable offer and launch an instance."""
query = f"num_gpus=1 rentable=True rented=False compute_cap>=700 dph<={MAX_COST_PER_HOUR}"
query = (
f"num_gpus=1 rentable=True rented=False compute_cap>=700"
f" reliability2>=0.95 inet_up>=100 dph<={MAX_COST_PER_HOUR}"
)
offers = self.sdk.search_offers(query=query, order="dph", limit=5)

if not offers:
Expand Down
155 changes: 155 additions & 0 deletions gpu_test/test_kernels.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@

from typing import TYPE_CHECKING

import numpy as np
import pytest

if TYPE_CHECKING:
Expand DownExpand Up@@ -583,3 +584,157 @@ def test_float_to_int_conversion(kernel_runner: KernelRunner) -> None:
forth_source=("\\! kernel main\n\\! param DATA i64[256]\n7.9 F>S\n0 CELLS DATA + !"),
)
assert result[0] == 7


# --- Attention ---

_ATTENTION_KERNEL = """\
\\! kernel attention
\\! param Q f64[{n}]
\\! param K f64[{n}]
\\! param V f64[{n}]
\\! param O f64[{n}]
\\! param SEQ_LEN i64
\\! param HEAD_DIM i64
\\! shared SCORES f64[{seq_len}]
\\! shared SCRATCH f64[{seq_len}]
BID-X
TID-X
0.0
HEAD_DIM 0 DO
2 PICK HEAD_DIM * I + CELLS Q + F@
2 PICK HEAD_DIM * I + CELLS K + F@
F* F+
LOOP
HEAD_DIM S>F FSQRT F/
OVER 3 PICK >
IF DROP -1.0e30 THEN
OVER CELLS SCORES + SF!
BARRIER
TID-X 0= IF
0 CELLS SCORES + SF@
SEQ_LEN 1 DO I CELLS SCORES + SF@ FMAX LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F- FEXP
OVER CELLS SCORES + SF!
BARRIER
TID-X 0= IF
0.0
SEQ_LEN 0 DO I CELLS SCORES + SF@ F+ LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F/
OVER CELLS SCORES + SF!
BARRIER
DUP BEGIN DUP HEAD_DIM < WHILE
0.0
SEQ_LEN 0 DO
I CELLS SCORES + SF@
I HEAD_DIM * 3 PICK + CELLS V + F@
F* F+
LOOP
OVER 4 PICK HEAD_DIM * + CELLS O + F!
BDIM-X +
REPEAT
DROP DROP DROP
"""


def _attention_reference(q: np.ndarray, k: np.ndarray, v: np.ndarray, seq_len: int) -> list[float]:
"""Compute scaled dot-product attention with causal mask (NumPy reference)."""
head_dim = q.shape[1]
scores = q @ k.T / np.sqrt(head_dim)
causal_mask = np.triu(np.ones((seq_len, seq_len), dtype=bool), k=1)
scores[causal_mask] = -1e30
exp_scores = np.exp(scores - scores.max(axis=1, keepdims=True))
attn = exp_scores / exp_scores.sum(axis=1, keepdims=True)
return (attn @ v).flatten().tolist()


def test_naive_attention_f64(kernel_runner: KernelRunner) -> None:
"""Naive scaled dot-product attention with causal mask.

O = softmax(Q @ K^T / sqrt(d_k)) @ V, seq_len=4, head_dim=4.
One block per query row, one thread per key position.
"""
seq_len, head_dim = 4, 4

q = np.array(
[
[1.0, 0.0, 1.0, 0.0],
[0.0, 1.0, 0.0, 1.0],
[1.0, 1.0, 0.0, 0.0],
[0.0, 0.0, 1.0, 1.0],
]
)
k = np.array(
[
[1.0, 0.0, 0.0, 1.0],
[0.0, 1.0, 1.0, 0.0],
[1.0, 1.0, 0.0, 0.0],
[0.0, 0.0, 1.0, 1.0],
]
)
v = np.array(
[
[1.0, 2.0, 3.0, 4.0],
[5.0, 6.0, 7.0, 8.0],
[9.0, 10.0, 11.0, 12.0],
[13.0, 14.0, 15.0, 16.0],
]
)

expected = _attention_reference(q, k, v, seq_len)
n = seq_len * head_dim

result = kernel_runner.run(
forth_source=_ATTENTION_KERNEL.format(n=n, seq_len=seq_len),
params={
"Q": q.flatten().tolist(),
"K": k.flatten().tolist(),
"V": v.flatten().tolist(),
"SEQ_LEN": seq_len,
"HEAD_DIM": head_dim,
},
grid=(seq_len, 1, 1),
block=(seq_len, 1, 1),
output_param=3,
output_count=n,
)
assert result == [pytest.approx(v) for v in expected]


def test_naive_attention_f64_16x64(kernel_runner: KernelRunner) -> None:
"""Naive scaled dot-product attention, seq_len=16, head_dim=64."""
seq_len, head_dim = 16, 64

rng = np.random.default_rng(42)
q = rng.standard_normal((seq_len, head_dim))
k = rng.standard_normal((seq_len, head_dim))
v = rng.standard_normal((seq_len, head_dim))

expected = _attention_reference(q, k, v, seq_len)
n = seq_len * head_dim

result = kernel_runner.run(
forth_source=_ATTENTION_KERNEL.format(n=n, seq_len=seq_len),
params={
"Q": q.flatten().tolist(),
"K": k.flatten().tolist(),
"V": v.flatten().tolist(),
"SEQ_LEN": seq_len,
"HEAD_DIM": head_dim,
},
grid=(seq_len, 1, 1),
block=(seq_len, 1, 1),
output_param=3,
output_count=n,
)
assert result == [pytest.approx(v) for v in expected]
5 changes: 5 additions & 0 deletions lib/Bitcode/CMakeLists.txt
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,5 @@
set(WARPFORTH_LIBDEVICE_PATH
"${CMAKE_CURRENT_BINARY_DIR}/libdevice.10.bc" PARENT_SCOPE)

configure_file(libdevice.10.bc
"${CMAKE_CURRENT_BINARY_DIR}/libdevice.10.bc" COPYONLY)
Binary file addedlib/Bitcode/libdevice.10.bc
Binary file not shown.
1 change: 1 addition & 0 deletions lib/CMakeLists.txt
Original file line numberDiff line numberDiff line change
@@ -1,3 +1,4 @@
add_subdirectory(Dialect)
add_subdirectory(Bitcode)
add_subdirectory(Conversion)
add_subdirectory(Translation)
4 changes: 4 additions & 0 deletions lib/Conversion/CMakeLists.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,5 +16,9 @@ add_mlir_library(MLIRConversionPasses
MLIRTransforms
)

target_compile_definitions(obj.MLIRConversionPasses PRIVATE
WARPFORTH_LIBDEVICE_PATH="${WARPFORTH_LIBDEVICE_PATH}"
)

add_subdirectory(ForthToMemRef)
add_subdirectory(ForthToGPU)
5 changes: 4 additions & 1 deletion lib/Conversion/Passes.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -33,7 +33,10 @@ void buildWarpForthPipeline(OpPassManager &pm) {
pm.addPass(createCanonicalizerPass());

// Stage 4: Attach NVVM target to GPU modules (sm_70 = Volta architecture)
pm.addPass(createGpuNVVMAttachTarget());
GpuNVVMAttachTargetOptions nvvmOptions;
nvvmOptions.chip = "sm_70";
nvvmOptions.linkLibs.push_back(WARPFORTH_LIBDEVICE_PATH);
pm.addPass(createGpuNVVMAttachTarget(nvvmOptions));

// Stage 5: Lower GPU to NVVM with bare pointers
ConvertGpuOpsToNVVMOpsOptions gpuToNVVMOptions;
Expand Down
1 change: 1 addition & 0 deletions pyproject.toml
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@ version = "0.1.0"
requires-python = ">=3.11"
dependencies = [
"lit>=18.1.0",
"numpy",
"pytest",
"vastai-sdk",
]
Expand Down
80 changes: 80 additions & 0 deletions test/Pipeline/attention.forth
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,80 @@
\ RUN: %warpforth-translate --forth-to-mlir %s | %warpforth-opt --warpforth-pipeline | %FileCheck %s

\ Verify that a naive attention kernel with shared memory and float intrinsics
\ survives the full pipeline to gpu.binary.
\ CHECK: gpu.binary @warpforth_module

\! kernel attention
\! param Q f64[16]
\! param K f64[16]
\! param V f64[16]
\! param O f64[16]
\! param SEQ_LEN i64
\! param HEAD_DIM i64
\! shared SCORES f64[4]
\! shared SCRATCH f64[4]

\ row = BID-X, t = TID-X
BID-X
TID-X

\ --- Dot product: Q[row,:] . K[t,:] ---
0.0
HEAD_DIM 0 DO
2 PICK HEAD_DIM * I + CELLS Q + F@
2 PICK HEAD_DIM * I + CELLS K + F@
F* F+
LOOP
HEAD_DIM S>F FSQRT F/

\ --- Causal mask: if t > row, score = -inf ---
OVER 3 PICK >
IF DROP -1.0e30 THEN

\ --- Store score to shared memory ---
OVER CELLS SCORES + SF!
BARRIER

\ --- Softmax: max reduction (thread 0) ---
TID-X 0= IF
0 CELLS SCORES + SF@
SEQ_LEN 1 DO I CELLS SCORES + SF@ FMAX LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER

\ --- Softmax: exp(score - max) ---
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F- FEXP
OVER CELLS SCORES + SF!
BARRIER

\ --- Softmax: sum reduction (thread 0) ---
TID-X 0= IF
0.0
SEQ_LEN 0 DO I CELLS SCORES + SF@ F+ LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER

\ --- Softmax: normalize ---
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F/
OVER CELLS SCORES + SF!
BARRIER

\ --- V accumulation: O[row,col] = sum_j SCORES[j] * V[j*HD + col] ---
\ Stride over head_dim columns: col = t, t+BDIM-X, t+2*BDIM-X, ...
DUP BEGIN DUP HEAD_DIM < WHILE
0.0
SEQ_LEN 0 DO
I CELLS SCORES + SF@
I HEAD_DIM * 3 PICK + CELLS V + F@
F* F+
LOOP
OVER 4 PICK HEAD_DIM * + CELLS O + F!
BDIM-X +
REPEAT
DROP DROP DROP
2 changes: 2 additions & 0 deletions uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion gpu_test/conftest.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -118,7 +118,10 @@ def _cleanup_orphans(self) -> None:

def _launch(self) -> None:
"""Find the cheapest suitable offer and launch an instance."""
query = f"num_gpus=1 rentable=True rented=False compute_cap>=700 dph<={MAX_COST_PER_HOUR}"
query = (
f"num_gpus=1 rentable=True rented=False compute_cap>=700"
f" reliability2>=0.95 inet_up>=100 dph<={MAX_COST_PER_HOUR}"
)
offers = self.sdk.search_offers(query=query, order="dph", limit=5)

if not offers:
Expand Down
155 changes: 155 additions & 0 deletions gpu_test/test_kernels.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@

from typing import TYPE_CHECKING

import numpy as np
import pytest

if TYPE_CHECKING:
Expand DownExpand Up@@ -583,3 +584,157 @@ def test_float_to_int_conversion(kernel_runner: KernelRunner) -> None:
forth_source=("\\! kernel main\n\\! param DATA i64[256]\n7.9 F>S\n0 CELLS DATA + !"),
)
assert result[0] == 7


# --- Attention ---

_ATTENTION_KERNEL = """\
\\! kernel attention
\\! param Q f64[{n}]
\\! param K f64[{n}]
\\! param V f64[{n}]
\\! param O f64[{n}]
\\! param SEQ_LEN i64
\\! param HEAD_DIM i64
\\! shared SCORES f64[{seq_len}]
\\! shared SCRATCH f64[{seq_len}]
BID-X
TID-X
0.0
HEAD_DIM 0 DO
2 PICK HEAD_DIM * I + CELLS Q + F@
2 PICK HEAD_DIM * I + CELLS K + F@
F* F+
LOOP
HEAD_DIM S>F FSQRT F/
OVER 3 PICK >
IF DROP -1.0e30 THEN
OVER CELLS SCORES + SF!
BARRIER
TID-X 0= IF
0 CELLS SCORES + SF@
SEQ_LEN 1 DO I CELLS SCORES + SF@ FMAX LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F- FEXP
OVER CELLS SCORES + SF!
BARRIER
TID-X 0= IF
0.0
SEQ_LEN 0 DO I CELLS SCORES + SF@ F+ LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F/
OVER CELLS SCORES + SF!
BARRIER
DUP BEGIN DUP HEAD_DIM < WHILE
0.0
SEQ_LEN 0 DO
I CELLS SCORES + SF@
I HEAD_DIM * 3 PICK + CELLS V + F@
F* F+
LOOP
OVER 4 PICK HEAD_DIM * + CELLS O + F!
BDIM-X +
REPEAT
DROP DROP DROP
"""


def _attention_reference(q: np.ndarray, k: np.ndarray, v: np.ndarray, seq_len: int) -> list[float]:
"""Compute scaled dot-product attention with causal mask (NumPy reference)."""
head_dim = q.shape[1]
scores = q @ k.T / np.sqrt(head_dim)
causal_mask = np.triu(np.ones((seq_len, seq_len), dtype=bool), k=1)
scores[causal_mask] = -1e30
exp_scores = np.exp(scores - scores.max(axis=1, keepdims=True))
attn = exp_scores / exp_scores.sum(axis=1, keepdims=True)
return (attn @ v).flatten().tolist()


def test_naive_attention_f64(kernel_runner: KernelRunner) -> None:
"""Naive scaled dot-product attention with causal mask.

O = softmax(Q @ K^T / sqrt(d_k)) @ V, seq_len=4, head_dim=4.
One block per query row, one thread per key position.
"""
seq_len, head_dim = 4, 4

q = np.array(
[
[1.0, 0.0, 1.0, 0.0],
[0.0, 1.0, 0.0, 1.0],
[1.0, 1.0, 0.0, 0.0],
[0.0, 0.0, 1.0, 1.0],
]
)
k = np.array(
[
[1.0, 0.0, 0.0, 1.0],
[0.0, 1.0, 1.0, 0.0],
[1.0, 1.0, 0.0, 0.0],
[0.0, 0.0, 1.0, 1.0],
]
)
v = np.array(
[
[1.0, 2.0, 3.0, 4.0],
[5.0, 6.0, 7.0, 8.0],
[9.0, 10.0, 11.0, 12.0],
[13.0, 14.0, 15.0, 16.0],
]
)

expected = _attention_reference(q, k, v, seq_len)
n = seq_len * head_dim

result = kernel_runner.run(
forth_source=_ATTENTION_KERNEL.format(n=n, seq_len=seq_len),
params={
"Q": q.flatten().tolist(),
"K": k.flatten().tolist(),
"V": v.flatten().tolist(),
"SEQ_LEN": seq_len,
"HEAD_DIM": head_dim,
},
grid=(seq_len, 1, 1),
block=(seq_len, 1, 1),
output_param=3,
output_count=n,
)
assert result == [pytest.approx(v) for v in expected]


def test_naive_attention_f64_16x64(kernel_runner: KernelRunner) -> None:
"""Naive scaled dot-product attention, seq_len=16, head_dim=64."""
seq_len, head_dim = 16, 64

rng = np.random.default_rng(42)
q = rng.standard_normal((seq_len, head_dim))
k = rng.standard_normal((seq_len, head_dim))
v = rng.standard_normal((seq_len, head_dim))

expected = _attention_reference(q, k, v, seq_len)
n = seq_len * head_dim

result = kernel_runner.run(
forth_source=_ATTENTION_KERNEL.format(n=n, seq_len=seq_len),
params={
"Q": q.flatten().tolist(),
"K": k.flatten().tolist(),
"V": v.flatten().tolist(),
"SEQ_LEN": seq_len,
"HEAD_DIM": head_dim,
},
grid=(seq_len, 1, 1),
block=(seq_len, 1, 1),
output_param=3,
output_count=n,
)
assert result == [pytest.approx(v) for v in expected]
5 changes: 5 additions & 0 deletions lib/Bitcode/CMakeLists.txt
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,5 @@
set(WARPFORTH_LIBDEVICE_PATH
"${CMAKE_CURRENT_BINARY_DIR}/libdevice.10.bc" PARENT_SCOPE)

configure_file(libdevice.10.bc
"${CMAKE_CURRENT_BINARY_DIR}/libdevice.10.bc" COPYONLY)
Binary file addedlib/Bitcode/libdevice.10.bc
Binary file not shown.
1 change: 1 addition & 0 deletions lib/CMakeLists.txt
Original file line numberDiff line numberDiff line change
@@ -1,3 +1,4 @@
add_subdirectory(Dialect)
add_subdirectory(Bitcode)
add_subdirectory(Conversion)
add_subdirectory(Translation)
4 changes: 4 additions & 0 deletions lib/Conversion/CMakeLists.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,5 +16,9 @@ add_mlir_library(MLIRConversionPasses
MLIRTransforms
)

target_compile_definitions(obj.MLIRConversionPasses PRIVATE
WARPFORTH_LIBDEVICE_PATH="${WARPFORTH_LIBDEVICE_PATH}"
)

add_subdirectory(ForthToMemRef)
add_subdirectory(ForthToGPU)
5 changes: 4 additions & 1 deletion lib/Conversion/Passes.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -33,7 +33,10 @@ void buildWarpForthPipeline(OpPassManager &pm) {
pm.addPass(createCanonicalizerPass());

// Stage 4: Attach NVVM target to GPU modules (sm_70 = Volta architecture)
pm.addPass(createGpuNVVMAttachTarget());
GpuNVVMAttachTargetOptions nvvmOptions;
nvvmOptions.chip = "sm_70";
nvvmOptions.linkLibs.push_back(WARPFORTH_LIBDEVICE_PATH);
pm.addPass(createGpuNVVMAttachTarget(nvvmOptions));

// Stage 5: Lower GPU to NVVM with bare pointers
ConvertGpuOpsToNVVMOpsOptions gpuToNVVMOptions;
Expand Down
1 change: 1 addition & 0 deletions pyproject.toml
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@ version = "0.1.0"
requires-python = ">=3.11"
dependencies = [
"lit>=18.1.0",
"numpy",
"pytest",
"vastai-sdk",
]
Expand Down
80 changes: 80 additions & 0 deletions test/Pipeline/attention.forth
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,80 @@
\ RUN: %warpforth-translate --forth-to-mlir %s | %warpforth-opt --warpforth-pipeline | %FileCheck %s

\ Verify that a naive attention kernel with shared memory and float intrinsics
\ survives the full pipeline to gpu.binary.
\ CHECK: gpu.binary @warpforth_module

\! kernel attention
\! param Q f64[16]
\! param K f64[16]
\! param V f64[16]
\! param O f64[16]
\! param SEQ_LEN i64
\! param HEAD_DIM i64
\! shared SCORES f64[4]
\! shared SCRATCH f64[4]

\ row = BID-X, t = TID-X
BID-X
TID-X

\ --- Dot product: Q[row,:] . K[t,:] ---
0.0
HEAD_DIM 0 DO
2 PICK HEAD_DIM * I + CELLS Q + F@
2 PICK HEAD_DIM * I + CELLS K + F@
F* F+
LOOP
HEAD_DIM S>F FSQRT F/

\ --- Causal mask: if t > row, score = -inf ---
OVER 3 PICK >
IF DROP -1.0e30 THEN

\ --- Store score to shared memory ---
OVER CELLS SCORES + SF!
BARRIER

\ --- Softmax: max reduction (thread 0) ---
TID-X 0= IF
0 CELLS SCORES + SF@
SEQ_LEN 1 DO I CELLS SCORES + SF@ FMAX LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER

\ --- Softmax: exp(score - max) ---
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F- FEXP
OVER CELLS SCORES + SF!
BARRIER

\ --- Softmax: sum reduction (thread 0) ---
TID-X 0= IF
0.0
SEQ_LEN 0 DO I CELLS SCORES + SF@ F+ LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER

\ --- Softmax: normalize ---
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F/
OVER CELLS SCORES + SF!
BARRIER

\ --- V accumulation: O[row,col] = sum_j SCORES[j] * V[j*HD + col] ---
\ Stride over head_dim columns: col = t, t+BDIM-X, t+2*BDIM-X, ...
DUP BEGIN DUP HEAD_DIM < WHILE
0.0
SEQ_LEN 0 DO
I CELLS SCORES + SF@
I HEAD_DIM * 3 PICK + CELLS V + F@
F* F+
LOOP
OVER 4 PICK HEAD_DIM * + CELLS O + F!
BDIM-X +
REPEAT
DROP DROP DROP
2 changes: 2 additions & 0 deletions uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion gpu_test/conftest.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -118,7 +118,10 @@ def _cleanup_orphans(self) -> None:

def _launch(self) -> None:
"""Find the cheapest suitable offer and launch an instance."""
query = f"num_gpus=1 rentable=True rented=False compute_cap>=700 dph<={MAX_COST_PER_HOUR}"
query = (
f"num_gpus=1 rentable=True rented=False compute_cap>=700"
f" reliability2>=0.95 inet_up>=100 dph<={MAX_COST_PER_HOUR}"
)
offers = self.sdk.search_offers(query=query, order="dph", limit=5)

if not offers:
Expand Down
155 changes: 155 additions & 0 deletions gpu_test/test_kernels.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@

from typing import TYPE_CHECKING

import numpy as np
import pytest

if TYPE_CHECKING:
Expand DownExpand Up@@ -583,3 +584,157 @@ def test_float_to_int_conversion(kernel_runner: KernelRunner) -> None:
forth_source=("\\! kernel main\n\\! param DATA i64[256]\n7.9 F>S\n0 CELLS DATA + !"),
)
assert result[0] == 7


# --- Attention ---

_ATTENTION_KERNEL = """\
\\! kernel attention
\\! param Q f64[{n}]
\\! param K f64[{n}]
\\! param V f64[{n}]
\\! param O f64[{n}]
\\! param SEQ_LEN i64
\\! param HEAD_DIM i64
\\! shared SCORES f64[{seq_len}]
\\! shared SCRATCH f64[{seq_len}]
BID-X
TID-X
0.0
HEAD_DIM 0 DO
2 PICK HEAD_DIM * I + CELLS Q + F@
2 PICK HEAD_DIM * I + CELLS K + F@
F* F+
LOOP
HEAD_DIM S>F FSQRT F/
OVER 3 PICK >
IF DROP -1.0e30 THEN
OVER CELLS SCORES + SF!
BARRIER
TID-X 0= IF
0 CELLS SCORES + SF@
SEQ_LEN 1 DO I CELLS SCORES + SF@ FMAX LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F- FEXP
OVER CELLS SCORES + SF!
BARRIER
TID-X 0= IF
0.0
SEQ_LEN 0 DO I CELLS SCORES + SF@ F+ LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F/
OVER CELLS SCORES + SF!
BARRIER
DUP BEGIN DUP HEAD_DIM < WHILE
0.0
SEQ_LEN 0 DO
I CELLS SCORES + SF@
I HEAD_DIM * 3 PICK + CELLS V + F@
F* F+
LOOP
OVER 4 PICK HEAD_DIM * + CELLS O + F!
BDIM-X +
REPEAT
DROP DROP DROP
"""


def _attention_reference(q: np.ndarray, k: np.ndarray, v: np.ndarray, seq_len: int) -> list[float]:
"""Compute scaled dot-product attention with causal mask (NumPy reference)."""
head_dim = q.shape[1]
scores = q @ k.T / np.sqrt(head_dim)
causal_mask = np.triu(np.ones((seq_len, seq_len), dtype=bool), k=1)
scores[causal_mask] = -1e30
exp_scores = np.exp(scores - scores.max(axis=1, keepdims=True))
attn = exp_scores / exp_scores.sum(axis=1, keepdims=True)
return (attn @ v).flatten().tolist()


def test_naive_attention_f64(kernel_runner: KernelRunner) -> None:
"""Naive scaled dot-product attention with causal mask.

O = softmax(Q @ K^T / sqrt(d_k)) @ V, seq_len=4, head_dim=4.
One block per query row, one thread per key position.
"""
seq_len, head_dim = 4, 4

q = np.array(
[
[1.0, 0.0, 1.0, 0.0],
[0.0, 1.0, 0.0, 1.0],
[1.0, 1.0, 0.0, 0.0],
[0.0, 0.0, 1.0, 1.0],
]
)
k = np.array(
[
[1.0, 0.0, 0.0, 1.0],
[0.0, 1.0, 1.0, 0.0],
[1.0, 1.0, 0.0, 0.0],
[0.0, 0.0, 1.0, 1.0],
]
)
v = np.array(
[
[1.0, 2.0, 3.0, 4.0],
[5.0, 6.0, 7.0, 8.0],
[9.0, 10.0, 11.0, 12.0],
[13.0, 14.0, 15.0, 16.0],
]
)

expected = _attention_reference(q, k, v, seq_len)
n = seq_len * head_dim

result = kernel_runner.run(
forth_source=_ATTENTION_KERNEL.format(n=n, seq_len=seq_len),
params={
"Q": q.flatten().tolist(),
"K": k.flatten().tolist(),
"V": v.flatten().tolist(),
"SEQ_LEN": seq_len,
"HEAD_DIM": head_dim,
},
grid=(seq_len, 1, 1),
block=(seq_len, 1, 1),
output_param=3,
output_count=n,
)
assert result == [pytest.approx(v) for v in expected]


def test_naive_attention_f64_16x64(kernel_runner: KernelRunner) -> None:
"""Naive scaled dot-product attention, seq_len=16, head_dim=64."""
seq_len, head_dim = 16, 64

rng = np.random.default_rng(42)
q = rng.standard_normal((seq_len, head_dim))
k = rng.standard_normal((seq_len, head_dim))
v = rng.standard_normal((seq_len, head_dim))

expected = _attention_reference(q, k, v, seq_len)
n = seq_len * head_dim

result = kernel_runner.run(
forth_source=_ATTENTION_KERNEL.format(n=n, seq_len=seq_len),
params={
"Q": q.flatten().tolist(),
"K": k.flatten().tolist(),
"V": v.flatten().tolist(),
"SEQ_LEN": seq_len,
"HEAD_DIM": head_dim,
},
grid=(seq_len, 1, 1),
block=(seq_len, 1, 1),
output_param=3,
output_count=n,
)
assert result == [pytest.approx(v) for v in expected]
5 changes: 5 additions & 0 deletions lib/Bitcode/CMakeLists.txt
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,5 @@
set(WARPFORTH_LIBDEVICE_PATH
"${CMAKE_CURRENT_BINARY_DIR}/libdevice.10.bc" PARENT_SCOPE)

configure_file(libdevice.10.bc
"${CMAKE_CURRENT_BINARY_DIR}/libdevice.10.bc" COPYONLY)
Binary file addedlib/Bitcode/libdevice.10.bc
Binary file not shown.
1 change: 1 addition & 0 deletions lib/CMakeLists.txt
Original file line numberDiff line numberDiff line change
@@ -1,3 +1,4 @@
add_subdirectory(Dialect)
add_subdirectory(Bitcode)
add_subdirectory(Conversion)
add_subdirectory(Translation)
4 changes: 4 additions & 0 deletions lib/Conversion/CMakeLists.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,5 +16,9 @@ add_mlir_library(MLIRConversionPasses
MLIRTransforms
)

target_compile_definitions(obj.MLIRConversionPasses PRIVATE
WARPFORTH_LIBDEVICE_PATH="${WARPFORTH_LIBDEVICE_PATH}"
)

add_subdirectory(ForthToMemRef)
add_subdirectory(ForthToGPU)
5 changes: 4 additions & 1 deletion lib/Conversion/Passes.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -33,7 +33,10 @@ void buildWarpForthPipeline(OpPassManager &pm) {
pm.addPass(createCanonicalizerPass());

// Stage 4: Attach NVVM target to GPU modules (sm_70 = Volta architecture)
pm.addPass(createGpuNVVMAttachTarget());
GpuNVVMAttachTargetOptions nvvmOptions;
nvvmOptions.chip = "sm_70";
nvvmOptions.linkLibs.push_back(WARPFORTH_LIBDEVICE_PATH);
pm.addPass(createGpuNVVMAttachTarget(nvvmOptions));

// Stage 5: Lower GPU to NVVM with bare pointers
ConvertGpuOpsToNVVMOpsOptions gpuToNVVMOptions;
Expand Down
1 change: 1 addition & 0 deletions pyproject.toml
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@ version = "0.1.0"
requires-python = ">=3.11"
dependencies = [
"lit>=18.1.0",
"numpy",
"pytest",
"vastai-sdk",
]
Expand Down
80 changes: 80 additions & 0 deletions test/Pipeline/attention.forth
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,80 @@
\ RUN: %warpforth-translate --forth-to-mlir %s | %warpforth-opt --warpforth-pipeline | %FileCheck %s

\ Verify that a naive attention kernel with shared memory and float intrinsics
\ survives the full pipeline to gpu.binary.
\ CHECK: gpu.binary @warpforth_module

\! kernel attention
\! param Q f64[16]
\! param K f64[16]
\! param V f64[16]
\! param O f64[16]
\! param SEQ_LEN i64
\! param HEAD_DIM i64
\! shared SCORES f64[4]
\! shared SCRATCH f64[4]

\ row = BID-X, t = TID-X
BID-X
TID-X

\ --- Dot product: Q[row,:] . K[t,:] ---
0.0
HEAD_DIM 0 DO
2 PICK HEAD_DIM * I + CELLS Q + F@
2 PICK HEAD_DIM * I + CELLS K + F@
F* F+
LOOP
HEAD_DIM S>F FSQRT F/

\ --- Causal mask: if t > row, score = -inf ---
OVER 3 PICK >
IF DROP -1.0e30 THEN

\ --- Store score to shared memory ---
OVER CELLS SCORES + SF!
BARRIER

\ --- Softmax: max reduction (thread 0) ---
TID-X 0= IF
0 CELLS SCORES + SF@
SEQ_LEN 1 DO I CELLS SCORES + SF@ FMAX LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER

\ --- Softmax: exp(score - max) ---
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F- FEXP
OVER CELLS SCORES + SF!
BARRIER

\ --- Softmax: sum reduction (thread 0) ---
TID-X 0= IF
0.0
SEQ_LEN 0 DO I CELLS SCORES + SF@ F+ LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER

\ --- Softmax: normalize ---
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F/
OVER CELLS SCORES + SF!
BARRIER

\ --- V accumulation: O[row,col] = sum_j SCORES[j] * V[j*HD + col] ---
\ Stride over head_dim columns: col = t, t+BDIM-X, t+2*BDIM-X, ...
DUP BEGIN DUP HEAD_DIM < WHILE
0.0
SEQ_LEN 0 DO
I CELLS SCORES + SF@
I HEAD_DIM * 3 PICK + CELLS V + F@
F* F+
LOOP
OVER 4 PICK HEAD_DIM * + CELLS O + F!
BDIM-X +
REPEAT
DROP DROP DROP
2 changes: 2 additions & 0 deletions uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion gpu_test/conftest.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -118,7 +118,10 @@ def _cleanup_orphans(self) -> None:

def _launch(self) -> None:
"""Find the cheapest suitable offer and launch an instance."""
query = f"num_gpus=1 rentable=True rented=False compute_cap>=700 dph<={MAX_COST_PER_HOUR}"
query = (
f"num_gpus=1 rentable=True rented=False compute_cap>=700"
f" reliability2>=0.95 inet_up>=100 dph<={MAX_COST_PER_HOUR}"
)
offers = self.sdk.search_offers(query=query, order="dph", limit=5)

if not offers:
Expand Down
155 changes: 155 additions & 0 deletions gpu_test/test_kernels.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@

from typing import TYPE_CHECKING

import numpy as np
import pytest

if TYPE_CHECKING:
Expand DownExpand Up@@ -583,3 +584,157 @@ def test_float_to_int_conversion(kernel_runner: KernelRunner) -> None:
forth_source=("\\! kernel main\n\\! param DATA i64[256]\n7.9 F>S\n0 CELLS DATA + !"),
)
assert result[0] == 7


# --- Attention ---

_ATTENTION_KERNEL = """\
\\! kernel attention
\\! param Q f64[{n}]
\\! param K f64[{n}]
\\! param V f64[{n}]
\\! param O f64[{n}]
\\! param SEQ_LEN i64
\\! param HEAD_DIM i64
\\! shared SCORES f64[{seq_len}]
\\! shared SCRATCH f64[{seq_len}]
BID-X
TID-X
0.0
HEAD_DIM 0 DO
2 PICK HEAD_DIM * I + CELLS Q + F@
2 PICK HEAD_DIM * I + CELLS K + F@
F* F+
LOOP
HEAD_DIM S>F FSQRT F/
OVER 3 PICK >
IF DROP -1.0e30 THEN
OVER CELLS SCORES + SF!
BARRIER
TID-X 0= IF
0 CELLS SCORES + SF@
SEQ_LEN 1 DO I CELLS SCORES + SF@ FMAX LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F- FEXP
OVER CELLS SCORES + SF!
BARRIER
TID-X 0= IF
0.0
SEQ_LEN 0 DO I CELLS SCORES + SF@ F+ LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F/
OVER CELLS SCORES + SF!
BARRIER
DUP BEGIN DUP HEAD_DIM < WHILE
0.0
SEQ_LEN 0 DO
I CELLS SCORES + SF@
I HEAD_DIM * 3 PICK + CELLS V + F@
F* F+
LOOP
OVER 4 PICK HEAD_DIM * + CELLS O + F!
BDIM-X +
REPEAT
DROP DROP DROP
"""


def _attention_reference(q: np.ndarray, k: np.ndarray, v: np.ndarray, seq_len: int) -> list[float]:
"""Compute scaled dot-product attention with causal mask (NumPy reference)."""
head_dim = q.shape[1]
scores = q @ k.T / np.sqrt(head_dim)
causal_mask = np.triu(np.ones((seq_len, seq_len), dtype=bool), k=1)
scores[causal_mask] = -1e30
exp_scores = np.exp(scores - scores.max(axis=1, keepdims=True))
attn = exp_scores / exp_scores.sum(axis=1, keepdims=True)
return (attn @ v).flatten().tolist()


def test_naive_attention_f64(kernel_runner: KernelRunner) -> None:
"""Naive scaled dot-product attention with causal mask.

O = softmax(Q @ K^T / sqrt(d_k)) @ V, seq_len=4, head_dim=4.
One block per query row, one thread per key position.
"""
seq_len, head_dim = 4, 4

q = np.array(
[
[1.0, 0.0, 1.0, 0.0],
[0.0, 1.0, 0.0, 1.0],
[1.0, 1.0, 0.0, 0.0],
[0.0, 0.0, 1.0, 1.0],
]
)
k = np.array(
[
[1.0, 0.0, 0.0, 1.0],
[0.0, 1.0, 1.0, 0.0],
[1.0, 1.0, 0.0, 0.0],
[0.0, 0.0, 1.0, 1.0],
]
)
v = np.array(
[
[1.0, 2.0, 3.0, 4.0],
[5.0, 6.0, 7.0, 8.0],
[9.0, 10.0, 11.0, 12.0],
[13.0, 14.0, 15.0, 16.0],
]
)

expected = _attention_reference(q, k, v, seq_len)
n = seq_len * head_dim

result = kernel_runner.run(
forth_source=_ATTENTION_KERNEL.format(n=n, seq_len=seq_len),
params={
"Q": q.flatten().tolist(),
"K": k.flatten().tolist(),
"V": v.flatten().tolist(),
"SEQ_LEN": seq_len,
"HEAD_DIM": head_dim,
},
grid=(seq_len, 1, 1),
block=(seq_len, 1, 1),
output_param=3,
output_count=n,
)
assert result == [pytest.approx(v) for v in expected]


def test_naive_attention_f64_16x64(kernel_runner: KernelRunner) -> None:
"""Naive scaled dot-product attention, seq_len=16, head_dim=64."""
seq_len, head_dim = 16, 64

rng = np.random.default_rng(42)
q = rng.standard_normal((seq_len, head_dim))
k = rng.standard_normal((seq_len, head_dim))
v = rng.standard_normal((seq_len, head_dim))

expected = _attention_reference(q, k, v, seq_len)
n = seq_len * head_dim

result = kernel_runner.run(
forth_source=_ATTENTION_KERNEL.format(n=n, seq_len=seq_len),
params={
"Q": q.flatten().tolist(),
"K": k.flatten().tolist(),
"V": v.flatten().tolist(),
"SEQ_LEN": seq_len,
"HEAD_DIM": head_dim,
},
grid=(seq_len, 1, 1),
block=(seq_len, 1, 1),
output_param=3,
output_count=n,
)
assert result == [pytest.approx(v) for v in expected]
5 changes: 5 additions & 0 deletions lib/Bitcode/CMakeLists.txt
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,5 @@
set(WARPFORTH_LIBDEVICE_PATH
"${CMAKE_CURRENT_BINARY_DIR}/libdevice.10.bc" PARENT_SCOPE)

configure_file(libdevice.10.bc
"${CMAKE_CURRENT_BINARY_DIR}/libdevice.10.bc" COPYONLY)
Binary file addedlib/Bitcode/libdevice.10.bc
Binary file not shown.
1 change: 1 addition & 0 deletions lib/CMakeLists.txt
Original file line numberDiff line numberDiff line change
@@ -1,3 +1,4 @@
add_subdirectory(Dialect)
add_subdirectory(Bitcode)
add_subdirectory(Conversion)
add_subdirectory(Translation)
4 changes: 4 additions & 0 deletions lib/Conversion/CMakeLists.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,5 +16,9 @@ add_mlir_library(MLIRConversionPasses
MLIRTransforms
)

target_compile_definitions(obj.MLIRConversionPasses PRIVATE
WARPFORTH_LIBDEVICE_PATH="${WARPFORTH_LIBDEVICE_PATH}"
)

add_subdirectory(ForthToMemRef)
add_subdirectory(ForthToGPU)
5 changes: 4 additions & 1 deletion lib/Conversion/Passes.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -33,7 +33,10 @@ void buildWarpForthPipeline(OpPassManager &pm) {
pm.addPass(createCanonicalizerPass());

// Stage 4: Attach NVVM target to GPU modules (sm_70 = Volta architecture)
pm.addPass(createGpuNVVMAttachTarget());
GpuNVVMAttachTargetOptions nvvmOptions;
nvvmOptions.chip = "sm_70";
nvvmOptions.linkLibs.push_back(WARPFORTH_LIBDEVICE_PATH);
pm.addPass(createGpuNVVMAttachTarget(nvvmOptions));

// Stage 5: Lower GPU to NVVM with bare pointers
ConvertGpuOpsToNVVMOpsOptions gpuToNVVMOptions;
Expand Down
1 change: 1 addition & 0 deletions pyproject.toml
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@ version = "0.1.0"
requires-python = ">=3.11"
dependencies = [
"lit>=18.1.0",
"numpy",
"pytest",
"vastai-sdk",
]
Expand Down
80 changes: 80 additions & 0 deletions test/Pipeline/attention.forth
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,80 @@
\ RUN: %warpforth-translate --forth-to-mlir %s | %warpforth-opt --warpforth-pipeline | %FileCheck %s

\ Verify that a naive attention kernel with shared memory and float intrinsics
\ survives the full pipeline to gpu.binary.
\ CHECK: gpu.binary @warpforth_module

\! kernel attention
\! param Q f64[16]
\! param K f64[16]
\! param V f64[16]
\! param O f64[16]
\! param SEQ_LEN i64
\! param HEAD_DIM i64
\! shared SCORES f64[4]
\! shared SCRATCH f64[4]

\ row = BID-X, t = TID-X
BID-X
TID-X

\ --- Dot product: Q[row,:] . K[t,:] ---
0.0
HEAD_DIM 0 DO
2 PICK HEAD_DIM * I + CELLS Q + F@
2 PICK HEAD_DIM * I + CELLS K + F@
F* F+
LOOP
HEAD_DIM S>F FSQRT F/

\ --- Causal mask: if t > row, score = -inf ---
OVER 3 PICK >
IF DROP -1.0e30 THEN

\ --- Store score to shared memory ---
OVER CELLS SCORES + SF!
BARRIER

\ --- Softmax: max reduction (thread 0) ---
TID-X 0= IF
0 CELLS SCORES + SF@
SEQ_LEN 1 DO I CELLS SCORES + SF@ FMAX LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER

\ --- Softmax: exp(score - max) ---
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F- FEXP
OVER CELLS SCORES + SF!
BARRIER

\ --- Softmax: sum reduction (thread 0) ---
TID-X 0= IF
0.0
SEQ_LEN 0 DO I CELLS SCORES + SF@ F+ LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER

\ --- Softmax: normalize ---
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F/
OVER CELLS SCORES + SF!
BARRIER

\ --- V accumulation: O[row,col] = sum_j SCORES[j] * V[j*HD + col] ---
\ Stride over head_dim columns: col = t, t+BDIM-X, t+2*BDIM-X, ...
DUP BEGIN DUP HEAD_DIM < WHILE
0.0
SEQ_LEN 0 DO
I CELLS SCORES + SF@
I HEAD_DIM * 3 PICK + CELLS V + F@
F* F+
LOOP
OVER 4 PICK HEAD_DIM * + CELLS O + F!
BDIM-X +
REPEAT
DROP DROP DROP
2 changes: 2 additions & 0 deletions uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion gpu_test/conftest.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -118,7 +118,10 @@ def _cleanup_orphans(self) -> None:

def _launch(self) -> None:
"""Find the cheapest suitable offer and launch an instance."""
query = f"num_gpus=1 rentable=True rented=False compute_cap>=700 dph<={MAX_COST_PER_HOUR}"
query = (
f"num_gpus=1 rentable=True rented=False compute_cap>=700"
f" reliability2>=0.95 inet_up>=100 dph<={MAX_COST_PER_HOUR}"
)
offers = self.sdk.search_offers(query=query, order="dph", limit=5)

if not offers:
Expand Down
155 changes: 155 additions & 0 deletions gpu_test/test_kernels.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@

from typing import TYPE_CHECKING

import numpy as np
import pytest

if TYPE_CHECKING:
Expand DownExpand Up@@ -583,3 +584,157 @@ def test_float_to_int_conversion(kernel_runner: KernelRunner) -> None:
forth_source=("\\! kernel main\n\\! param DATA i64[256]\n7.9 F>S\n0 CELLS DATA + !"),
)
assert result[0] == 7


# --- Attention ---

_ATTENTION_KERNEL = """\
\\! kernel attention
\\! param Q f64[{n}]
\\! param K f64[{n}]
\\! param V f64[{n}]
\\! param O f64[{n}]
\\! param SEQ_LEN i64
\\! param HEAD_DIM i64
\\! shared SCORES f64[{seq_len}]
\\! shared SCRATCH f64[{seq_len}]
BID-X
TID-X
0.0
HEAD_DIM 0 DO
2 PICK HEAD_DIM * I + CELLS Q + F@
2 PICK HEAD_DIM * I + CELLS K + F@
F* F+
LOOP
HEAD_DIM S>F FSQRT F/
OVER 3 PICK >
IF DROP -1.0e30 THEN
OVER CELLS SCORES + SF!
BARRIER
TID-X 0= IF
0 CELLS SCORES + SF@
SEQ_LEN 1 DO I CELLS SCORES + SF@ FMAX LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F- FEXP
OVER CELLS SCORES + SF!
BARRIER
TID-X 0= IF
0.0
SEQ_LEN 0 DO I CELLS SCORES + SF@ F+ LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F/
OVER CELLS SCORES + SF!
BARRIER
DUP BEGIN DUP HEAD_DIM < WHILE
0.0
SEQ_LEN 0 DO
I CELLS SCORES + SF@
I HEAD_DIM * 3 PICK + CELLS V + F@
F* F+
LOOP
OVER 4 PICK HEAD_DIM * + CELLS O + F!
BDIM-X +
REPEAT
DROP DROP DROP
"""


def _attention_reference(q: np.ndarray, k: np.ndarray, v: np.ndarray, seq_len: int) -> list[float]:
"""Compute scaled dot-product attention with causal mask (NumPy reference)."""
head_dim = q.shape[1]
scores = q @ k.T / np.sqrt(head_dim)
causal_mask = np.triu(np.ones((seq_len, seq_len), dtype=bool), k=1)
scores[causal_mask] = -1e30
exp_scores = np.exp(scores - scores.max(axis=1, keepdims=True))
attn = exp_scores / exp_scores.sum(axis=1, keepdims=True)
return (attn @ v).flatten().tolist()


def test_naive_attention_f64(kernel_runner: KernelRunner) -> None:
"""Naive scaled dot-product attention with causal mask.

O = softmax(Q @ K^T / sqrt(d_k)) @ V, seq_len=4, head_dim=4.
One block per query row, one thread per key position.
"""
seq_len, head_dim = 4, 4

q = np.array(
[
[1.0, 0.0, 1.0, 0.0],
[0.0, 1.0, 0.0, 1.0],
[1.0, 1.0, 0.0, 0.0],
[0.0, 0.0, 1.0, 1.0],
]
)
k = np.array(
[
[1.0, 0.0, 0.0, 1.0],
[0.0, 1.0, 1.0, 0.0],
[1.0, 1.0, 0.0, 0.0],
[0.0, 0.0, 1.0, 1.0],
]
)
v = np.array(
[
[1.0, 2.0, 3.0, 4.0],
[5.0, 6.0, 7.0, 8.0],
[9.0, 10.0, 11.0, 12.0],
[13.0, 14.0, 15.0, 16.0],
]
)

expected = _attention_reference(q, k, v, seq_len)
n = seq_len * head_dim

result = kernel_runner.run(
forth_source=_ATTENTION_KERNEL.format(n=n, seq_len=seq_len),
params={
"Q": q.flatten().tolist(),
"K": k.flatten().tolist(),
"V": v.flatten().tolist(),
"SEQ_LEN": seq_len,
"HEAD_DIM": head_dim,
},
grid=(seq_len, 1, 1),
block=(seq_len, 1, 1),
output_param=3,
output_count=n,
)
assert result == [pytest.approx(v) for v in expected]


def test_naive_attention_f64_16x64(kernel_runner: KernelRunner) -> None:
"""Naive scaled dot-product attention, seq_len=16, head_dim=64."""
seq_len, head_dim = 16, 64

rng = np.random.default_rng(42)
q = rng.standard_normal((seq_len, head_dim))
k = rng.standard_normal((seq_len, head_dim))
v = rng.standard_normal((seq_len, head_dim))

expected = _attention_reference(q, k, v, seq_len)
n = seq_len * head_dim

result = kernel_runner.run(
forth_source=_ATTENTION_KERNEL.format(n=n, seq_len=seq_len),
params={
"Q": q.flatten().tolist(),
"K": k.flatten().tolist(),
"V": v.flatten().tolist(),
"SEQ_LEN": seq_len,
"HEAD_DIM": head_dim,
},
grid=(seq_len, 1, 1),
block=(seq_len, 1, 1),
output_param=3,
output_count=n,
)
assert result == [pytest.approx(v) for v in expected]
5 changes: 5 additions & 0 deletions lib/Bitcode/CMakeLists.txt
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,5 @@
set(WARPFORTH_LIBDEVICE_PATH
"${CMAKE_CURRENT_BINARY_DIR}/libdevice.10.bc" PARENT_SCOPE)

configure_file(libdevice.10.bc
"${CMAKE_CURRENT_BINARY_DIR}/libdevice.10.bc" COPYONLY)
Binary file addedlib/Bitcode/libdevice.10.bc
Binary file not shown.
1 change: 1 addition & 0 deletions lib/CMakeLists.txt
Original file line numberDiff line numberDiff line change
@@ -1,3 +1,4 @@
add_subdirectory(Dialect)
add_subdirectory(Bitcode)
add_subdirectory(Conversion)
add_subdirectory(Translation)
4 changes: 4 additions & 0 deletions lib/Conversion/CMakeLists.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,5 +16,9 @@ add_mlir_library(MLIRConversionPasses
MLIRTransforms
)

target_compile_definitions(obj.MLIRConversionPasses PRIVATE
WARPFORTH_LIBDEVICE_PATH="${WARPFORTH_LIBDEVICE_PATH}"
)

add_subdirectory(ForthToMemRef)
add_subdirectory(ForthToGPU)
5 changes: 4 additions & 1 deletion lib/Conversion/Passes.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -33,7 +33,10 @@ void buildWarpForthPipeline(OpPassManager &pm) {
pm.addPass(createCanonicalizerPass());

// Stage 4: Attach NVVM target to GPU modules (sm_70 = Volta architecture)
pm.addPass(createGpuNVVMAttachTarget());
GpuNVVMAttachTargetOptions nvvmOptions;
nvvmOptions.chip = "sm_70";
nvvmOptions.linkLibs.push_back(WARPFORTH_LIBDEVICE_PATH);
pm.addPass(createGpuNVVMAttachTarget(nvvmOptions));

// Stage 5: Lower GPU to NVVM with bare pointers
ConvertGpuOpsToNVVMOpsOptions gpuToNVVMOptions;
Expand Down
1 change: 1 addition & 0 deletions pyproject.toml
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@ version = "0.1.0"
requires-python = ">=3.11"
dependencies = [
"lit>=18.1.0",
"numpy",
"pytest",
"vastai-sdk",
]
Expand Down
80 changes: 80 additions & 0 deletions test/Pipeline/attention.forth
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,80 @@
\ RUN: %warpforth-translate --forth-to-mlir %s | %warpforth-opt --warpforth-pipeline | %FileCheck %s

\ Verify that a naive attention kernel with shared memory and float intrinsics
\ survives the full pipeline to gpu.binary.
\ CHECK: gpu.binary @warpforth_module

\! kernel attention
\! param Q f64[16]
\! param K f64[16]
\! param V f64[16]
\! param O f64[16]
\! param SEQ_LEN i64
\! param HEAD_DIM i64
\! shared SCORES f64[4]
\! shared SCRATCH f64[4]

\ row = BID-X, t = TID-X
BID-X
TID-X

\ --- Dot product: Q[row,:] . K[t,:] ---
0.0
HEAD_DIM 0 DO
2 PICK HEAD_DIM * I + CELLS Q + F@
2 PICK HEAD_DIM * I + CELLS K + F@
F* F+
LOOP
HEAD_DIM S>F FSQRT F/

\ --- Causal mask: if t > row, score = -inf ---
OVER 3 PICK >
IF DROP -1.0e30 THEN

\ --- Store score to shared memory ---
OVER CELLS SCORES + SF!
BARRIER

\ --- Softmax: max reduction (thread 0) ---
TID-X 0= IF
0 CELLS SCORES + SF@
SEQ_LEN 1 DO I CELLS SCORES + SF@ FMAX LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER

\ --- Softmax: exp(score - max) ---
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F- FEXP
OVER CELLS SCORES + SF!
BARRIER

\ --- Softmax: sum reduction (thread 0) ---
TID-X 0= IF
0.0
SEQ_LEN 0 DO I CELLS SCORES + SF@ F+ LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER

\ --- Softmax: normalize ---
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F/
OVER CELLS SCORES + SF!
BARRIER

\ --- V accumulation: O[row,col] = sum_j SCORES[j] * V[j*HD + col] ---
\ Stride over head_dim columns: col = t, t+BDIM-X, t+2*BDIM-X, ...
DUP BEGIN DUP HEAD_DIM < WHILE
0.0
SEQ_LEN 0 DO
I CELLS SCORES + SF@
I HEAD_DIM * 3 PICK + CELLS V + F@
F* F+
LOOP
OVER 4 PICK HEAD_DIM * + CELLS O + F!
BDIM-X +
REPEAT
DROP DROP DROP
2 changes: 2 additions & 0 deletions uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
5 changes: 4 additions & 1 deletion gpu_test/conftest.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -118,7 +118,10 @@ def _cleanup_orphans(self) -> None:

def _launch(self) -> None:
"""Find the cheapest suitable offer and launch an instance."""
query = f"num_gpus=1 rentable=True rented=False compute_cap>=700 dph<={MAX_COST_PER_HOUR}"
query = (
f"num_gpus=1 rentable=True rented=False compute_cap>=700"
f" reliability2>=0.95 inet_up>=100 dph<={MAX_COST_PER_HOUR}"
)
offers = self.sdk.search_offers(query=query, order="dph", limit=5)

if not offers:
Expand Down
155 changes: 155 additions & 0 deletions gpu_test/test_kernels.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@

from typing import TYPE_CHECKING

import numpy as np
import pytest

if TYPE_CHECKING:
Expand DownExpand Up@@ -583,3 +584,157 @@ def test_float_to_int_conversion(kernel_runner: KernelRunner) -> None:
forth_source=("\\! kernel main\n\\! param DATA i64[256]\n7.9 F>S\n0 CELLS DATA + !"),
)
assert result[0] == 7


# --- Attention ---

_ATTENTION_KERNEL = """\
\\! kernel attention
\\! param Q f64[{n}]
\\! param K f64[{n}]
\\! param V f64[{n}]
\\! param O f64[{n}]
\\! param SEQ_LEN i64
\\! param HEAD_DIM i64
\\! shared SCORES f64[{seq_len}]
\\! shared SCRATCH f64[{seq_len}]
BID-X
TID-X
0.0
HEAD_DIM 0 DO
2 PICK HEAD_DIM * I + CELLS Q + F@
2 PICK HEAD_DIM * I + CELLS K + F@
F* F+
LOOP
HEAD_DIM S>F FSQRT F/
OVER 3 PICK >
IF DROP -1.0e30 THEN
OVER CELLS SCORES + SF!
BARRIER
TID-X 0= IF
0 CELLS SCORES + SF@
SEQ_LEN 1 DO I CELLS SCORES + SF@ FMAX LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F- FEXP
OVER CELLS SCORES + SF!
BARRIER
TID-X 0= IF
0.0
SEQ_LEN 0 DO I CELLS SCORES + SF@ F+ LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F/
OVER CELLS SCORES + SF!
BARRIER
DUP BEGIN DUP HEAD_DIM < WHILE
0.0
SEQ_LEN 0 DO
I CELLS SCORES + SF@
I HEAD_DIM * 3 PICK + CELLS V + F@
F* F+
LOOP
OVER 4 PICK HEAD_DIM * + CELLS O + F!
BDIM-X +
REPEAT
DROP DROP DROP
"""


def _attention_reference(q: np.ndarray, k: np.ndarray, v: np.ndarray, seq_len: int) -> list[float]:
"""Compute scaled dot-product attention with causal mask (NumPy reference)."""
head_dim = q.shape[1]
scores = q @ k.T / np.sqrt(head_dim)
causal_mask = np.triu(np.ones((seq_len, seq_len), dtype=bool), k=1)
scores[causal_mask] = -1e30
exp_scores = np.exp(scores - scores.max(axis=1, keepdims=True))
attn = exp_scores / exp_scores.sum(axis=1, keepdims=True)
return (attn @ v).flatten().tolist()


def test_naive_attention_f64(kernel_runner: KernelRunner) -> None:
"""Naive scaled dot-product attention with causal mask.

O = softmax(Q @ K^T / sqrt(d_k)) @ V, seq_len=4, head_dim=4.
One block per query row, one thread per key position.
"""
seq_len, head_dim = 4, 4

q = np.array(
[
[1.0, 0.0, 1.0, 0.0],
[0.0, 1.0, 0.0, 1.0],
[1.0, 1.0, 0.0, 0.0],
[0.0, 0.0, 1.0, 1.0],
]
)
k = np.array(
[
[1.0, 0.0, 0.0, 1.0],
[0.0, 1.0, 1.0, 0.0],
[1.0, 1.0, 0.0, 0.0],
[0.0, 0.0, 1.0, 1.0],
]
)
v = np.array(
[
[1.0, 2.0, 3.0, 4.0],
[5.0, 6.0, 7.0, 8.0],
[9.0, 10.0, 11.0, 12.0],
[13.0, 14.0, 15.0, 16.0],
]
)

expected = _attention_reference(q, k, v, seq_len)
n = seq_len * head_dim

result = kernel_runner.run(
forth_source=_ATTENTION_KERNEL.format(n=n, seq_len=seq_len),
params={
"Q": q.flatten().tolist(),
"K": k.flatten().tolist(),
"V": v.flatten().tolist(),
"SEQ_LEN": seq_len,
"HEAD_DIM": head_dim,
},
grid=(seq_len, 1, 1),
block=(seq_len, 1, 1),
output_param=3,
output_count=n,
)
assert result == [pytest.approx(v) for v in expected]


def test_naive_attention_f64_16x64(kernel_runner: KernelRunner) -> None:
"""Naive scaled dot-product attention, seq_len=16, head_dim=64."""
seq_len, head_dim = 16, 64

rng = np.random.default_rng(42)
q = rng.standard_normal((seq_len, head_dim))
k = rng.standard_normal((seq_len, head_dim))
v = rng.standard_normal((seq_len, head_dim))

expected = _attention_reference(q, k, v, seq_len)
n = seq_len * head_dim

result = kernel_runner.run(
forth_source=_ATTENTION_KERNEL.format(n=n, seq_len=seq_len),
params={
"Q": q.flatten().tolist(),
"K": k.flatten().tolist(),
"V": v.flatten().tolist(),
"SEQ_LEN": seq_len,
"HEAD_DIM": head_dim,
},
grid=(seq_len, 1, 1),
block=(seq_len, 1, 1),
output_param=3,
output_count=n,
)
assert result == [pytest.approx(v) for v in expected]
5 changes: 5 additions & 0 deletions lib/Bitcode/CMakeLists.txt
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,5 @@
set(WARPFORTH_LIBDEVICE_PATH
"${CMAKE_CURRENT_BINARY_DIR}/libdevice.10.bc" PARENT_SCOPE)

configure_file(libdevice.10.bc
"${CMAKE_CURRENT_BINARY_DIR}/libdevice.10.bc" COPYONLY)
Binary file addedlib/Bitcode/libdevice.10.bc
Binary file not shown.
1 change: 1 addition & 0 deletions lib/CMakeLists.txt
Original file line numberDiff line numberDiff line change
@@ -1,3 +1,4 @@
add_subdirectory(Dialect)
add_subdirectory(Bitcode)
add_subdirectory(Conversion)
add_subdirectory(Translation)
4 changes: 4 additions & 0 deletions lib/Conversion/CMakeLists.txt
Original file line numberDiff line numberDiff line change
Expand Up@@ -16,5 +16,9 @@ add_mlir_library(MLIRConversionPasses
MLIRTransforms
)

target_compile_definitions(obj.MLIRConversionPasses PRIVATE
WARPFORTH_LIBDEVICE_PATH="${WARPFORTH_LIBDEVICE_PATH}"
)

add_subdirectory(ForthToMemRef)
add_subdirectory(ForthToGPU)
5 changes: 4 additions & 1 deletion lib/Conversion/Passes.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -33,7 +33,10 @@ void buildWarpForthPipeline(OpPassManager &pm) {
pm.addPass(createCanonicalizerPass());

// Stage 4: Attach NVVM target to GPU modules (sm_70 = Volta architecture)
pm.addPass(createGpuNVVMAttachTarget());
GpuNVVMAttachTargetOptions nvvmOptions;
nvvmOptions.chip = "sm_70";
nvvmOptions.linkLibs.push_back(WARPFORTH_LIBDEVICE_PATH);
pm.addPass(createGpuNVVMAttachTarget(nvvmOptions));

// Stage 5: Lower GPU to NVVM with bare pointers
ConvertGpuOpsToNVVMOpsOptions gpuToNVVMOptions;
Expand Down
1 change: 1 addition & 0 deletions pyproject.toml
Original file line numberDiff line numberDiff line change
Expand Up@@ -4,6 +4,7 @@ version = "0.1.0"
requires-python = ">=3.11"
dependencies = [
"lit>=18.1.0",
"numpy",
"pytest",
"vastai-sdk",
]
Expand Down
80 changes: 80 additions & 0 deletions test/Pipeline/attention.forth
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,80 @@
\ RUN: %warpforth-translate --forth-to-mlir %s | %warpforth-opt --warpforth-pipeline | %FileCheck %s

\ Verify that a naive attention kernel with shared memory and float intrinsics
\ survives the full pipeline to gpu.binary.
\ CHECK: gpu.binary @warpforth_module

\! kernel attention
\! param Q f64[16]
\! param K f64[16]
\! param V f64[16]
\! param O f64[16]
\! param SEQ_LEN i64
\! param HEAD_DIM i64
\! shared SCORES f64[4]
\! shared SCRATCH f64[4]

\ row = BID-X, t = TID-X
BID-X
TID-X

\ --- Dot product: Q[row,:] . K[t,:] ---
0.0
HEAD_DIM 0 DO
2 PICK HEAD_DIM * I + CELLS Q + F@
2 PICK HEAD_DIM * I + CELLS K + F@
F* F+
LOOP
HEAD_DIM S>F FSQRT F/

\ --- Causal mask: if t > row, score = -inf ---
OVER 3 PICK >
IF DROP -1.0e30 THEN

\ --- Store score to shared memory ---
OVER CELLS SCORES + SF!
BARRIER

\ --- Softmax: max reduction (thread 0) ---
TID-X 0= IF
0 CELLS SCORES + SF@
SEQ_LEN 1 DO I CELLS SCORES + SF@ FMAX LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER

\ --- Softmax: exp(score - max) ---
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F- FEXP
OVER CELLS SCORES + SF!
BARRIER

\ --- Softmax: sum reduction (thread 0) ---
TID-X 0= IF
0.0
SEQ_LEN 0 DO I CELLS SCORES + SF@ F+ LOOP
0 CELLS SCRATCH + SF!
THEN
BARRIER

\ --- Softmax: normalize ---
DUP CELLS SCORES + SF@
0 CELLS SCRATCH + SF@
F/
OVER CELLS SCORES + SF!
BARRIER

\ --- V accumulation: O[row,col] = sum_j SCORES[j] * V[j*HD + col] ---
\ Stride over head_dim columns: col = t, t+BDIM-X, t+2*BDIM-X, ...
DUP BEGIN DUP HEAD_DIM < WHILE
0.0
SEQ_LEN 0 DO
I CELLS SCORES + SF@
I HEAD_DIM * 3 PICK + CELLS V + F@
F* F+
LOOP
OVER 4 PICK HEAD_DIM * + CELLS O + F!
BDIM-X +
REPEAT
DROP DROP DROP
2 changes: 2 additions & 0 deletions uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.