Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
137 changes: 128 additions & 9 deletions src/coreclr/jit/codegenwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -2597,8 +2597,15 @@ void CodeGen::genCodeForLclFld(GenTreeLclFld* tree)
NYI_WASM_SIMD("SIMD16 local field load");
}

GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(tree), tree->GetLclNum(), tree->GetLclOffs());
if (type == TYP_SIMD12)
{
genLoadLclTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(tree), tree->GetLclNum(), tree->GetLclOffs());
}
WasmProduceReg(tree);
}

Expand All@@ -2621,8 +2628,15 @@ void CodeGen::genCodeForLclVar(GenTreeLclVar* tree)
{
var_types type = varDsc->GetRegisterType(tree);

GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(type), tree->GetLclNum(), 0);
if (type == TYP_SIMD12)
{
genLoadLclTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(type), tree->GetLclNum(), 0);
}
WasmProduceReg(tree);
}
else
Expand DownExpand Up@@ -2695,6 +2709,94 @@ void CodeGen::genCodeForFrameSize(GenTree* tree)
WasmProduceReg(tree);
}

//------------------------------------------------------------------------
// genLoadLclTypeSimd12: Load a TYP_SIMD12 (i.e. Vector3) local into a v128.
//
// Arguments:
// tree - the GT_LCL_FLD or GT_LCL_VAR node
//
// Notes:
// Vector3 has no native wasm valtype, so it lives as a v128 with the low 12 bytes
// populated. The frame address is pushed twice: v128.load64_zero fills lanes 0-1
// (zeroing the rest) and v128.load32_lane fills lane 2 from bytes 8-11.
//
void CodeGen::genLoadLclTypeSimd12(GenTreeLclVarCommon* tree)
{
bool fpBased;
int frameOffset = m_compiler->lvaFrameAddress(tree->GetLclNum(), &fpBased) + (int)tree->GetLclOffs();
noway_assert(frameOffset >= 0); // WASM address modes are unsigned.
assert(fpBased);
unsigned fpIndex = GetFramePointerRegIndex();
emitter* emit = GetEmitter();

emit->emitIns_I(INS_local_get, EA_PTRSIZE, fpIndex);
emit->emitIns_I(INS_local_get, EA_PTRSIZE, fpIndex);
emit->emitIns_I(INS_v128_load64_zero, EA_8BYTE, frameOffset);
emit->emitIns_MemargLane(INS_v128_load32_lane, EA_4BYTE, frameOffset + 8, 2);
}

//------------------------------------------------------------------------
// genLoadIndTypeSimd12: Load a TYP_SIMD12 (i.e. Vector3) value through an indirection.
//
// Arguments:
// tree - the GT_IND node
//
// Notes:
// The address is left on the value stack by prior codegen and is multiply-used, so the
// trailing v128.load32_lane can re-push it for the upper 4 bytes.
//
void CodeGen::genLoadIndTypeSimd12(GenTreeIndir* tree)
{
emitter* emit = GetEmitter();

emit->emitIns_I(INS_local_get, EA_PTRSIZE, WasmRegToIndex(GetMultiUseOperandReg(tree->Addr())));
emit->emitIns_I(INS_v128_load64_zero, EA_8BYTE, 0);
emit->emitIns_MemargLane(INS_v128_load32_lane, EA_4BYTE, 8, 2);
}

//------------------------------------------------------------------------
// genStoreIndTypeSimd12: Store a TYP_SIMD12 (i.e. Vector3) value through an indirection.
//
// Arguments:
// tree - the GT_STOREIND node
//
// Notes:
// On entry the value stack holds [addr, value]. The value is teed into an internal v128
// local so it survives the low-8 store; the address is then re-materialized to store the
// upper 4 bytes via a lane store - re-emitting the frame pointer for a LCL_ADDR, or
// re-pushing the multiply-used address register otherwise.
//
void CodeGen::genStoreIndTypeSimd12(GenTreeStoreInd* tree)
{
emitter* emit = GetEmitter();
GenTree* addr = tree->Addr();

InternalRegs* regs = internalRegisters.GetAll(tree);
assert(regs->Count() == 1);
regNumber valReg = regs->Extract();

emit->emitIns_I(INS_local_tee, EA_16BYTE, WasmRegToIndex(valReg)); // [addr, value]
emit->emitIns_MemargLane(INS_v128_store64_lane, EA_8BYTE, 0, 0); // []

if (addr->OperIs(GT_LCL_ADDR))
{
bool fpBased;
int frameOffset = m_compiler->lvaFrameAddress(addr->AsLclVarCommon()->GetLclNum(), &fpBased) +
(int)addr->AsLclVarCommon()->GetLclOffs();
noway_assert(frameOffset >= 0); // WASM address modes are unsigned.
assert(fpBased);
emit->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex()); // [fp]
emit->emitIns_I(INS_local_get, EA_16BYTE, WasmRegToIndex(valReg)); // [fp, value]
emit->emitIns_MemargLane(INS_v128_store32_lane, EA_4BYTE, frameOffset + 8, 2); // []
}
else
{
emit->emitIns_I(INS_local_get, EA_PTRSIZE, WasmRegToIndex(GetMultiUseOperandReg(addr))); // [addr]
emit->emitIns_I(INS_local_get, EA_16BYTE, WasmRegToIndex(valReg)); // [addr, value]
emit->emitIns_MemargLane(INS_v128_store32_lane, EA_4BYTE, 8, 2); // []
}
}

//------------------------------------------------------------------------
// genCodeForIndir: Produce code for a GT_IND node.
//
Expand All@@ -2705,8 +2807,7 @@ void CodeGen::genCodeForIndir(GenTreeIndir* tree)
{
assert(tree->OperIs(GT_IND));

var_types type = tree->TypeGet();
instruction ins = ins_Load(type);
var_types type = tree->TypeGet();

genConsumeAddress(tree->Addr());

Expand All@@ -2718,7 +2819,14 @@ void CodeGen::genCodeForIndir(GenTreeIndir* tree)

// TODO-WASM: Memory barriers

GetEmitter()->emitIns_I(ins, emitActualTypeSize(type), 0);
if (type == TYP_SIMD12)
{
genLoadIndTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(ins_Load(type), emitActualTypeSize(type), 0);
}

WasmProduceReg(tree);
}
Expand DownExpand Up@@ -2762,11 +2870,22 @@ void CodeGen::genCodeForStoreInd(GenTreeStoreInd* tree)
// module. Bail until SIMD16 store is properly supported.
NYI_WASM_SIMD("SIMD16 store indirect");
}
instruction ins = ins_Store(type);

// TODO-WASM: Memory barriers

GetEmitter()->emitIns_I(ins, emitActualTypeSize(type), 0);
if (type == TYP_SIMD8)
{
// stack: [addr, value] -> store the low 8 bytes.
GetEmitter()->emitIns_MemargLane(INS_v128_store64_lane, EA_8BYTE, 0, 0);
}
else if (type == TYP_SIMD12)
{
genStoreIndTypeSimd12(tree);
}
Comment thread
tannergooding marked this conversation as resolved.
else
{
GetEmitter()->emitIns_I(ins_Store(type), emitActualTypeSize(type), 0);
}
}

genUpdateLife(tree);
Expand Down
4 changes: 4 additions & 0 deletions src/coreclr/jit/instr.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -2070,6 +2070,10 @@ instruction CodeGenInterface::ins_Load(var_types srcType, bool aligned /*=false*
case TYP_DOUBLE:
return INS_f64_load;
#if defined(FEATURE_SIMD)
case TYP_SIMD8:
// SIMD8 (Vector2) lives as a v128 with the low 8 bytes populated. SIMD12 (Vector3) is
// handled at the callers since it needs a trailing lane load for the upper 4 bytes.
return INS_v128_load64_zero;
case TYP_SIMD16:
return INS_v128_load;
#endif
Expand Down
14 changes: 10 additions & 4 deletions src/coreclr/jit/lowerwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -164,10 +164,14 @@ GenTree* Lowering::LowerStoreLoc(GenTreeLclVarCommon* storeLoc)
//
GenTree* Lowering::LowerStoreIndir(GenTreeStoreInd* node)
{
if ((node->gtFlags & GTF_IND_NONFAULTING) == 0)
if (((node->gtFlags & GTF_IND_NONFAULTING) == 0) ||
(node->TypeIs(TYP_SIMD12) && !node->Addr()->OperIs(GT_LCL_ADDR)))
{
// We need to be able to null check the address, and that requires multiple uses of the address operand.
SetMultiplyUsed(node->Addr() DEBUGARG("LowerStoreIndir faulting Addr"));
// SIMD12 stores also re-materialize the address for the trailing lane store, so force it there as well -
// unless the address is a re-materializable LCL_ADDR (the local-to-stack store rewrite), which codegen
// re-emits directly.
SetMultiplyUsed(node->Addr() DEBUGARG("LowerStoreIndir Addr (null check or simd12 lane store)"));
}

ContainCheckStoreIndir(node);
Expand DownExpand Up@@ -459,9 +463,11 @@ void Lowering::ContainCheckIndir(GenTreeIndir* indirNode)
return;
}

if (indirNode->OperIs(GT_IND) && ((indirNode->gtFlags & GTF_IND_NONFAULTING) == 0))
if (indirNode->OperIs(GT_IND) &&
(((indirNode->gtFlags & GTF_IND_NONFAULTING) == 0) || indirNode->TypeIs(TYP_SIMD12)))
{
SetMultiplyUsed(indirNode->Addr() DEBUGARG("ContainCheckIndir faulting load Addr"));
// SIMD12 loads re-materialize the address for the trailing lane load, so force it there regardless.
SetMultiplyUsed(indirNode->Addr() DEBUGARG("ContainCheckIndir load Addr (null check or simd12 lane load)"));
}

// TODO-WASM-CQ: contain suitable LEAs here. Take note of the fact that for this to be correct we must prove the
Expand Down
18 changes: 18 additions & 0 deletions src/coreclr/jit/regallocwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -662,6 +662,14 @@ void WasmRegAlloc::CollectReferencesForIndir(GenTreeIndir* node)
{
GenTree* const addr = node->Addr();
ConsumeTemporaryRegForOperand(addr DEBUGARG("indirection address"));

if (node->OperIs(GT_STOREIND) && node->TypeIs(TYP_SIMD12))
{
// The SIMD12 store stashes the v128 value so it can re-push it for the trailing lane store.
regNumber internalReg = RequestInternalRegister(node, TYP_SIMD16);
regNumber releasedReg = ReleaseTemporaryRegister(WasmRegToType(internalReg));
assert(releasedReg == internalReg);
}
}

//------------------------------------------------------------------------
Expand DownExpand Up@@ -800,6 +808,16 @@ void WasmRegAlloc::RewriteLocalStackStore(GenTreeLclVarCommon* lclNode)
LIR::ReadOnlyRange storeRange(store, store);
m_compiler->GetLowering()->LowerRange(m_currentBlock, storeRange);

if (store->OperIs(GT_STOREIND) && store->TypeIs(TYP_SIMD12))
{
// genStoreIndTypeSimd12 tees the value into a v128 temporary to split the store into an 8-byte and a
// 4-byte lane store. The main collection walk does not revisit this freshly-introduced node, so request
// that internal register here. The re-materializable LCL_ADDR address needs no temporary.
regNumber internalReg = RequestInternalRegister(store, TYP_SIMD16);
regNumber releasedReg = ReleaseTemporaryRegister(WasmRegToType(internalReg));
assert(releasedReg == internalReg);
}

// FIXME-WASM: Should we be doing this here?
// CollectReferencesForNode(store);
}
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Add copy buttons to all
 blocks
(function() {
function addCopyButtons() {
document.querySelectorAll('pre code').forEach(function(codeBlock) {
if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;
codeBlock.parentElement.setAttribute('data-copy-added', 'true');
var btn = document.createElement('button');
btn.textContent = 'Copy';
btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';
btn.onmouseover = function() { this.style.opacity = '1'; };
btn.onmouseout = function() { this.style.opacity = '0.7'; };
btn.onclick = function() {
navigator.clipboard.writeText(codeBlock.textContent).then(function() {
btn.textContent = 'Copied!';
setTimeout(function() { btn.textContent = 'Copy'; }, 1500);
});
};
codeBlock.parentElement.style.position = 'relative';
codeBlock.parentElement.appendChild(btn);
});
}
addCopyButtons();
// Re-run on dynamic content
var observer = new MutationObserver(addCopyButtons);
observer.observe(document.body, { childList: true, subtree: true });
})();
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
137 changes: 128 additions & 9 deletions src/coreclr/jit/codegenwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -2597,8 +2597,15 @@ void CodeGen::genCodeForLclFld(GenTreeLclFld* tree)
NYI_WASM_SIMD("SIMD16 local field load");
}

GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(tree), tree->GetLclNum(), tree->GetLclOffs());
if (type == TYP_SIMD12)
{
genLoadLclTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(tree), tree->GetLclNum(), tree->GetLclOffs());
}
WasmProduceReg(tree);
}

Expand All@@ -2621,8 +2628,15 @@ void CodeGen::genCodeForLclVar(GenTreeLclVar* tree)
{
var_types type = varDsc->GetRegisterType(tree);

GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(type), tree->GetLclNum(), 0);
if (type == TYP_SIMD12)
{
genLoadLclTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(type), tree->GetLclNum(), 0);
}
WasmProduceReg(tree);
}
else
Expand DownExpand Up@@ -2695,6 +2709,94 @@ void CodeGen::genCodeForFrameSize(GenTree* tree)
WasmProduceReg(tree);
}

//------------------------------------------------------------------------
// genLoadLclTypeSimd12: Load a TYP_SIMD12 (i.e. Vector3) local into a v128.
//
// Arguments:
// tree - the GT_LCL_FLD or GT_LCL_VAR node
//
// Notes:
// Vector3 has no native wasm valtype, so it lives as a v128 with the low 12 bytes
// populated. The frame address is pushed twice: v128.load64_zero fills lanes 0-1
// (zeroing the rest) and v128.load32_lane fills lane 2 from bytes 8-11.
//
void CodeGen::genLoadLclTypeSimd12(GenTreeLclVarCommon* tree)
{
bool fpBased;
int frameOffset = m_compiler->lvaFrameAddress(tree->GetLclNum(), &fpBased) + (int)tree->GetLclOffs();
noway_assert(frameOffset >= 0); // WASM address modes are unsigned.
assert(fpBased);
unsigned fpIndex = GetFramePointerRegIndex();
emitter* emit = GetEmitter();

emit->emitIns_I(INS_local_get, EA_PTRSIZE, fpIndex);
emit->emitIns_I(INS_local_get, EA_PTRSIZE, fpIndex);
emit->emitIns_I(INS_v128_load64_zero, EA_8BYTE, frameOffset);
emit->emitIns_MemargLane(INS_v128_load32_lane, EA_4BYTE, frameOffset + 8, 2);
}

//------------------------------------------------------------------------
// genLoadIndTypeSimd12: Load a TYP_SIMD12 (i.e. Vector3) value through an indirection.
//
// Arguments:
// tree - the GT_IND node
//
// Notes:
// The address is left on the value stack by prior codegen and is multiply-used, so the
// trailing v128.load32_lane can re-push it for the upper 4 bytes.
//
void CodeGen::genLoadIndTypeSimd12(GenTreeIndir* tree)
{
emitter* emit = GetEmitter();

emit->emitIns_I(INS_local_get, EA_PTRSIZE, WasmRegToIndex(GetMultiUseOperandReg(tree->Addr())));
emit->emitIns_I(INS_v128_load64_zero, EA_8BYTE, 0);
emit->emitIns_MemargLane(INS_v128_load32_lane, EA_4BYTE, 8, 2);
}

//------------------------------------------------------------------------
// genStoreIndTypeSimd12: Store a TYP_SIMD12 (i.e. Vector3) value through an indirection.
//
// Arguments:
// tree - the GT_STOREIND node
//
// Notes:
// On entry the value stack holds [addr, value]. The value is teed into an internal v128
// local so it survives the low-8 store; the address is then re-materialized to store the
// upper 4 bytes via a lane store - re-emitting the frame pointer for a LCL_ADDR, or
// re-pushing the multiply-used address register otherwise.
//
void CodeGen::genStoreIndTypeSimd12(GenTreeStoreInd* tree)
{
emitter* emit = GetEmitter();
GenTree* addr = tree->Addr();

InternalRegs* regs = internalRegisters.GetAll(tree);
assert(regs->Count() == 1);
regNumber valReg = regs->Extract();

emit->emitIns_I(INS_local_tee, EA_16BYTE, WasmRegToIndex(valReg)); // [addr, value]
emit->emitIns_MemargLane(INS_v128_store64_lane, EA_8BYTE, 0, 0); // []

if (addr->OperIs(GT_LCL_ADDR))
{
bool fpBased;
int frameOffset = m_compiler->lvaFrameAddress(addr->AsLclVarCommon()->GetLclNum(), &fpBased) +
(int)addr->AsLclVarCommon()->GetLclOffs();
noway_assert(frameOffset >= 0); // WASM address modes are unsigned.
assert(fpBased);
emit->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex()); // [fp]
emit->emitIns_I(INS_local_get, EA_16BYTE, WasmRegToIndex(valReg)); // [fp, value]
emit->emitIns_MemargLane(INS_v128_store32_lane, EA_4BYTE, frameOffset + 8, 2); // []
}
else
{
emit->emitIns_I(INS_local_get, EA_PTRSIZE, WasmRegToIndex(GetMultiUseOperandReg(addr))); // [addr]
emit->emitIns_I(INS_local_get, EA_16BYTE, WasmRegToIndex(valReg)); // [addr, value]
emit->emitIns_MemargLane(INS_v128_store32_lane, EA_4BYTE, 8, 2); // []
}
}

//------------------------------------------------------------------------
// genCodeForIndir: Produce code for a GT_IND node.
//
Expand All@@ -2705,8 +2807,7 @@ void CodeGen::genCodeForIndir(GenTreeIndir* tree)
{
assert(tree->OperIs(GT_IND));

var_types type = tree->TypeGet();
instruction ins = ins_Load(type);
var_types type = tree->TypeGet();

genConsumeAddress(tree->Addr());

Expand All@@ -2718,7 +2819,14 @@ void CodeGen::genCodeForIndir(GenTreeIndir* tree)

// TODO-WASM: Memory barriers

GetEmitter()->emitIns_I(ins, emitActualTypeSize(type), 0);
if (type == TYP_SIMD12)
{
genLoadIndTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(ins_Load(type), emitActualTypeSize(type), 0);
}

WasmProduceReg(tree);
}
Expand DownExpand Up@@ -2762,11 +2870,22 @@ void CodeGen::genCodeForStoreInd(GenTreeStoreInd* tree)
// module. Bail until SIMD16 store is properly supported.
NYI_WASM_SIMD("SIMD16 store indirect");
}
instruction ins = ins_Store(type);

// TODO-WASM: Memory barriers

GetEmitter()->emitIns_I(ins, emitActualTypeSize(type), 0);
if (type == TYP_SIMD8)
{
// stack: [addr, value] -> store the low 8 bytes.
GetEmitter()->emitIns_MemargLane(INS_v128_store64_lane, EA_8BYTE, 0, 0);
}
else if (type == TYP_SIMD12)
{
genStoreIndTypeSimd12(tree);
}
Comment thread
tannergooding marked this conversation as resolved.
else
{
GetEmitter()->emitIns_I(ins_Store(type), emitActualTypeSize(type), 0);
}
}

genUpdateLife(tree);
Expand Down
4 changes: 4 additions & 0 deletions src/coreclr/jit/instr.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -2070,6 +2070,10 @@ instruction CodeGenInterface::ins_Load(var_types srcType, bool aligned /*=false*
case TYP_DOUBLE:
return INS_f64_load;
#if defined(FEATURE_SIMD)
case TYP_SIMD8:
// SIMD8 (Vector2) lives as a v128 with the low 8 bytes populated. SIMD12 (Vector3) is
// handled at the callers since it needs a trailing lane load for the upper 4 bytes.
return INS_v128_load64_zero;
case TYP_SIMD16:
return INS_v128_load;
#endif
Expand Down
14 changes: 10 additions & 4 deletions src/coreclr/jit/lowerwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -164,10 +164,14 @@ GenTree* Lowering::LowerStoreLoc(GenTreeLclVarCommon* storeLoc)
//
GenTree* Lowering::LowerStoreIndir(GenTreeStoreInd* node)
{
if ((node->gtFlags & GTF_IND_NONFAULTING) == 0)
if (((node->gtFlags & GTF_IND_NONFAULTING) == 0) ||
(node->TypeIs(TYP_SIMD12) && !node->Addr()->OperIs(GT_LCL_ADDR)))
{
// We need to be able to null check the address, and that requires multiple uses of the address operand.
SetMultiplyUsed(node->Addr() DEBUGARG("LowerStoreIndir faulting Addr"));
// SIMD12 stores also re-materialize the address for the trailing lane store, so force it there as well -
// unless the address is a re-materializable LCL_ADDR (the local-to-stack store rewrite), which codegen
// re-emits directly.
SetMultiplyUsed(node->Addr() DEBUGARG("LowerStoreIndir Addr (null check or simd12 lane store)"));
}

ContainCheckStoreIndir(node);
Expand DownExpand Up@@ -459,9 +463,11 @@ void Lowering::ContainCheckIndir(GenTreeIndir* indirNode)
return;
}

if (indirNode->OperIs(GT_IND) && ((indirNode->gtFlags & GTF_IND_NONFAULTING) == 0))
if (indirNode->OperIs(GT_IND) &&
(((indirNode->gtFlags & GTF_IND_NONFAULTING) == 0) || indirNode->TypeIs(TYP_SIMD12)))
{
SetMultiplyUsed(indirNode->Addr() DEBUGARG("ContainCheckIndir faulting load Addr"));
// SIMD12 loads re-materialize the address for the trailing lane load, so force it there regardless.
SetMultiplyUsed(indirNode->Addr() DEBUGARG("ContainCheckIndir load Addr (null check or simd12 lane load)"));
}

// TODO-WASM-CQ: contain suitable LEAs here. Take note of the fact that for this to be correct we must prove the
Expand Down
18 changes: 18 additions & 0 deletions src/coreclr/jit/regallocwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -662,6 +662,14 @@ void WasmRegAlloc::CollectReferencesForIndir(GenTreeIndir* node)
{
GenTree* const addr = node->Addr();
ConsumeTemporaryRegForOperand(addr DEBUGARG("indirection address"));

if (node->OperIs(GT_STOREIND) && node->TypeIs(TYP_SIMD12))
{
// The SIMD12 store stashes the v128 value so it can re-push it for the trailing lane store.
regNumber internalReg = RequestInternalRegister(node, TYP_SIMD16);
regNumber releasedReg = ReleaseTemporaryRegister(WasmRegToType(internalReg));
assert(releasedReg == internalReg);
}
}

//------------------------------------------------------------------------
Expand DownExpand Up@@ -800,6 +808,16 @@ void WasmRegAlloc::RewriteLocalStackStore(GenTreeLclVarCommon* lclNode)
LIR::ReadOnlyRange storeRange(store, store);
m_compiler->GetLowering()->LowerRange(m_currentBlock, storeRange);

if (store->OperIs(GT_STOREIND) && store->TypeIs(TYP_SIMD12))
{
// genStoreIndTypeSimd12 tees the value into a v128 temporary to split the store into an 8-byte and a
// 4-byte lane store. The main collection walk does not revisit this freshly-introduced node, so request
// that internal register here. The re-materializable LCL_ADDR address needs no temporary.
regNumber internalReg = RequestInternalRegister(store, TYP_SIMD16);
regNumber releasedReg = ReleaseTemporaryRegister(WasmRegToType(internalReg));
assert(releasedReg == internalReg);
}

// FIXME-WASM: Should we be doing this here?
// CollectReferencesForNode(store);
}
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Force GitHub README to respect dark mode (function() { var style = document.createElement('style'); style.textContent = ' .markdown-body { color-scheme: dark light; } .markdown-body pre { background: #161b22 !important; } .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; } .markdown-body table th, .markdown-body table td { border-color: #30363d !important; } .markdown-body img { background: #0d1117; } .markdown-body blockquote { border-left-color: #8b949e; } .markdown-body hr { border-color: #30363d; } '; document.head.appendChild(style); })(); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
137 changes: 128 additions & 9 deletions src/coreclr/jit/codegenwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -2597,8 +2597,15 @@ void CodeGen::genCodeForLclFld(GenTreeLclFld* tree)
NYI_WASM_SIMD("SIMD16 local field load");
}

GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(tree), tree->GetLclNum(), tree->GetLclOffs());
if (type == TYP_SIMD12)
{
genLoadLclTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(tree), tree->GetLclNum(), tree->GetLclOffs());
}
WasmProduceReg(tree);
}

Expand All@@ -2621,8 +2628,15 @@ void CodeGen::genCodeForLclVar(GenTreeLclVar* tree)
{
var_types type = varDsc->GetRegisterType(tree);

GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(type), tree->GetLclNum(), 0);
if (type == TYP_SIMD12)
{
genLoadLclTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(type), tree->GetLclNum(), 0);
}
WasmProduceReg(tree);
}
else
Expand DownExpand Up@@ -2695,6 +2709,94 @@ void CodeGen::genCodeForFrameSize(GenTree* tree)
WasmProduceReg(tree);
}

//------------------------------------------------------------------------
// genLoadLclTypeSimd12: Load a TYP_SIMD12 (i.e. Vector3) local into a v128.
//
// Arguments:
// tree - the GT_LCL_FLD or GT_LCL_VAR node
//
// Notes:
// Vector3 has no native wasm valtype, so it lives as a v128 with the low 12 bytes
// populated. The frame address is pushed twice: v128.load64_zero fills lanes 0-1
// (zeroing the rest) and v128.load32_lane fills lane 2 from bytes 8-11.
//
void CodeGen::genLoadLclTypeSimd12(GenTreeLclVarCommon* tree)
{
bool fpBased;
int frameOffset = m_compiler->lvaFrameAddress(tree->GetLclNum(), &fpBased) + (int)tree->GetLclOffs();
noway_assert(frameOffset >= 0); // WASM address modes are unsigned.
assert(fpBased);
unsigned fpIndex = GetFramePointerRegIndex();
emitter* emit = GetEmitter();

emit->emitIns_I(INS_local_get, EA_PTRSIZE, fpIndex);
emit->emitIns_I(INS_local_get, EA_PTRSIZE, fpIndex);
emit->emitIns_I(INS_v128_load64_zero, EA_8BYTE, frameOffset);
emit->emitIns_MemargLane(INS_v128_load32_lane, EA_4BYTE, frameOffset + 8, 2);
}

//------------------------------------------------------------------------
// genLoadIndTypeSimd12: Load a TYP_SIMD12 (i.e. Vector3) value through an indirection.
//
// Arguments:
// tree - the GT_IND node
//
// Notes:
// The address is left on the value stack by prior codegen and is multiply-used, so the
// trailing v128.load32_lane can re-push it for the upper 4 bytes.
//
void CodeGen::genLoadIndTypeSimd12(GenTreeIndir* tree)
{
emitter* emit = GetEmitter();

emit->emitIns_I(INS_local_get, EA_PTRSIZE, WasmRegToIndex(GetMultiUseOperandReg(tree->Addr())));
emit->emitIns_I(INS_v128_load64_zero, EA_8BYTE, 0);
emit->emitIns_MemargLane(INS_v128_load32_lane, EA_4BYTE, 8, 2);
}

//------------------------------------------------------------------------
// genStoreIndTypeSimd12: Store a TYP_SIMD12 (i.e. Vector3) value through an indirection.
//
// Arguments:
// tree - the GT_STOREIND node
//
// Notes:
// On entry the value stack holds [addr, value]. The value is teed into an internal v128
// local so it survives the low-8 store; the address is then re-materialized to store the
// upper 4 bytes via a lane store - re-emitting the frame pointer for a LCL_ADDR, or
// re-pushing the multiply-used address register otherwise.
//
void CodeGen::genStoreIndTypeSimd12(GenTreeStoreInd* tree)
{
emitter* emit = GetEmitter();
GenTree* addr = tree->Addr();

InternalRegs* regs = internalRegisters.GetAll(tree);
assert(regs->Count() == 1);
regNumber valReg = regs->Extract();

emit->emitIns_I(INS_local_tee, EA_16BYTE, WasmRegToIndex(valReg)); // [addr, value]
emit->emitIns_MemargLane(INS_v128_store64_lane, EA_8BYTE, 0, 0); // []

if (addr->OperIs(GT_LCL_ADDR))
{
bool fpBased;
int frameOffset = m_compiler->lvaFrameAddress(addr->AsLclVarCommon()->GetLclNum(), &fpBased) +
(int)addr->AsLclVarCommon()->GetLclOffs();
noway_assert(frameOffset >= 0); // WASM address modes are unsigned.
assert(fpBased);
emit->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex()); // [fp]
emit->emitIns_I(INS_local_get, EA_16BYTE, WasmRegToIndex(valReg)); // [fp, value]
emit->emitIns_MemargLane(INS_v128_store32_lane, EA_4BYTE, frameOffset + 8, 2); // []
}
else
{
emit->emitIns_I(INS_local_get, EA_PTRSIZE, WasmRegToIndex(GetMultiUseOperandReg(addr))); // [addr]
emit->emitIns_I(INS_local_get, EA_16BYTE, WasmRegToIndex(valReg)); // [addr, value]
emit->emitIns_MemargLane(INS_v128_store32_lane, EA_4BYTE, 8, 2); // []
}
}

//------------------------------------------------------------------------
// genCodeForIndir: Produce code for a GT_IND node.
//
Expand All@@ -2705,8 +2807,7 @@ void CodeGen::genCodeForIndir(GenTreeIndir* tree)
{
assert(tree->OperIs(GT_IND));

var_types type = tree->TypeGet();
instruction ins = ins_Load(type);
var_types type = tree->TypeGet();

genConsumeAddress(tree->Addr());

Expand All@@ -2718,7 +2819,14 @@ void CodeGen::genCodeForIndir(GenTreeIndir* tree)

// TODO-WASM: Memory barriers

GetEmitter()->emitIns_I(ins, emitActualTypeSize(type), 0);
if (type == TYP_SIMD12)
{
genLoadIndTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(ins_Load(type), emitActualTypeSize(type), 0);
}

WasmProduceReg(tree);
}
Expand DownExpand Up@@ -2762,11 +2870,22 @@ void CodeGen::genCodeForStoreInd(GenTreeStoreInd* tree)
// module. Bail until SIMD16 store is properly supported.
NYI_WASM_SIMD("SIMD16 store indirect");
}
instruction ins = ins_Store(type);

// TODO-WASM: Memory barriers

GetEmitter()->emitIns_I(ins, emitActualTypeSize(type), 0);
if (type == TYP_SIMD8)
{
// stack: [addr, value] -> store the low 8 bytes.
GetEmitter()->emitIns_MemargLane(INS_v128_store64_lane, EA_8BYTE, 0, 0);
}
else if (type == TYP_SIMD12)
{
genStoreIndTypeSimd12(tree);
}
Comment thread
tannergooding marked this conversation as resolved.
else
{
GetEmitter()->emitIns_I(ins_Store(type), emitActualTypeSize(type), 0);
}
}

genUpdateLife(tree);
Expand Down
4 changes: 4 additions & 0 deletions src/coreclr/jit/instr.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -2070,6 +2070,10 @@ instruction CodeGenInterface::ins_Load(var_types srcType, bool aligned /*=false*
case TYP_DOUBLE:
return INS_f64_load;
#if defined(FEATURE_SIMD)
case TYP_SIMD8:
// SIMD8 (Vector2) lives as a v128 with the low 8 bytes populated. SIMD12 (Vector3) is
// handled at the callers since it needs a trailing lane load for the upper 4 bytes.
return INS_v128_load64_zero;
case TYP_SIMD16:
return INS_v128_load;
#endif
Expand Down
14 changes: 10 additions & 4 deletions src/coreclr/jit/lowerwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -164,10 +164,14 @@ GenTree* Lowering::LowerStoreLoc(GenTreeLclVarCommon* storeLoc)
//
GenTree* Lowering::LowerStoreIndir(GenTreeStoreInd* node)
{
if ((node->gtFlags & GTF_IND_NONFAULTING) == 0)
if (((node->gtFlags & GTF_IND_NONFAULTING) == 0) ||
(node->TypeIs(TYP_SIMD12) && !node->Addr()->OperIs(GT_LCL_ADDR)))
{
// We need to be able to null check the address, and that requires multiple uses of the address operand.
SetMultiplyUsed(node->Addr() DEBUGARG("LowerStoreIndir faulting Addr"));
// SIMD12 stores also re-materialize the address for the trailing lane store, so force it there as well -
// unless the address is a re-materializable LCL_ADDR (the local-to-stack store rewrite), which codegen
// re-emits directly.
SetMultiplyUsed(node->Addr() DEBUGARG("LowerStoreIndir Addr (null check or simd12 lane store)"));
}

ContainCheckStoreIndir(node);
Expand DownExpand Up@@ -459,9 +463,11 @@ void Lowering::ContainCheckIndir(GenTreeIndir* indirNode)
return;
}

if (indirNode->OperIs(GT_IND) && ((indirNode->gtFlags & GTF_IND_NONFAULTING) == 0))
if (indirNode->OperIs(GT_IND) &&
(((indirNode->gtFlags & GTF_IND_NONFAULTING) == 0) || indirNode->TypeIs(TYP_SIMD12)))
{
SetMultiplyUsed(indirNode->Addr() DEBUGARG("ContainCheckIndir faulting load Addr"));
// SIMD12 loads re-materialize the address for the trailing lane load, so force it there regardless.
SetMultiplyUsed(indirNode->Addr() DEBUGARG("ContainCheckIndir load Addr (null check or simd12 lane load)"));
}

// TODO-WASM-CQ: contain suitable LEAs here. Take note of the fact that for this to be correct we must prove the
Expand Down
18 changes: 18 additions & 0 deletions src/coreclr/jit/regallocwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -662,6 +662,14 @@ void WasmRegAlloc::CollectReferencesForIndir(GenTreeIndir* node)
{
GenTree* const addr = node->Addr();
ConsumeTemporaryRegForOperand(addr DEBUGARG("indirection address"));

if (node->OperIs(GT_STOREIND) && node->TypeIs(TYP_SIMD12))
{
// The SIMD12 store stashes the v128 value so it can re-push it for the trailing lane store.
regNumber internalReg = RequestInternalRegister(node, TYP_SIMD16);
regNumber releasedReg = ReleaseTemporaryRegister(WasmRegToType(internalReg));
assert(releasedReg == internalReg);
}
}

//------------------------------------------------------------------------
Expand DownExpand Up@@ -800,6 +808,16 @@ void WasmRegAlloc::RewriteLocalStackStore(GenTreeLclVarCommon* lclNode)
LIR::ReadOnlyRange storeRange(store, store);
m_compiler->GetLowering()->LowerRange(m_currentBlock, storeRange);

if (store->OperIs(GT_STOREIND) && store->TypeIs(TYP_SIMD12))
{
// genStoreIndTypeSimd12 tees the value into a v128 temporary to split the store into an 8-byte and a
// 4-byte lane store. The main collection walk does not revisit this freshly-introduced node, so request
// that internal register here. The re-materializable LCL_ADDR address needs no temporary.
regNumber internalReg = RequestInternalRegister(store, TYP_SIMD16);
regNumber releasedReg = ReleaseTemporaryRegister(WasmRegToType(internalReg));
assert(releasedReg == internalReg);
}

// FIXME-WASM: Should we be doing this here?
// CollectReferencesForNode(store);
}
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Highlight search terms from Google/DuckDuckGo/Bing referrer (function() { var ref = document.referrer; var terms = []; if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) { var url = new URL(ref); var q = url.searchParams.get('q') || url.searchParams.get('p'); if (q) { terms = q.split(/\s+/).filter(function(t) { return t.length > 2; }); } } if (terms.length === 0) return; var style = document.createElement('style'); style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }'; document.head.appendChild(style); function highlight(node) { if (node.nodeType === 3) { // text node var text = node.textContent; var found = false; terms.forEach(function(term) { var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\]\\]/g, '\\') + ')', 'gi'); if (regex.test(text)) { found = true; var frag = document.createDocumentFragment(); var parts = text.split(regex); parts.forEach(function(part, i) { if (i % 2 === 0) { frag.appendChild(document.createTextNode(part)); } else { var span = document.createElement('span'); span.className = 'userscript-highlight'; span.textContent = part; frag.appendChild(span); } }); node.parentNode.replaceChild(frag, node); } }); } else if (node.nodeType === 1 && node.childNodes) { // element var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT']; if (!skipTags.includes(node.tagName)) { Array.from(node.childNodes).forEach(highlight); } } } highlight(document.body); // Re-highlight on dynamic content var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1 || node.nodeType === 3) highlight(node); }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
137 changes: 128 additions & 9 deletions src/coreclr/jit/codegenwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -2597,8 +2597,15 @@ void CodeGen::genCodeForLclFld(GenTreeLclFld* tree)
NYI_WASM_SIMD("SIMD16 local field load");
}

GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(tree), tree->GetLclNum(), tree->GetLclOffs());
if (type == TYP_SIMD12)
{
genLoadLclTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(tree), tree->GetLclNum(), tree->GetLclOffs());
}
WasmProduceReg(tree);
}

Expand All@@ -2621,8 +2628,15 @@ void CodeGen::genCodeForLclVar(GenTreeLclVar* tree)
{
var_types type = varDsc->GetRegisterType(tree);

GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(type), tree->GetLclNum(), 0);
if (type == TYP_SIMD12)
{
genLoadLclTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(type), tree->GetLclNum(), 0);
}
WasmProduceReg(tree);
}
else
Expand DownExpand Up@@ -2695,6 +2709,94 @@ void CodeGen::genCodeForFrameSize(GenTree* tree)
WasmProduceReg(tree);
}

//------------------------------------------------------------------------
// genLoadLclTypeSimd12: Load a TYP_SIMD12 (i.e. Vector3) local into a v128.
//
// Arguments:
// tree - the GT_LCL_FLD or GT_LCL_VAR node
//
// Notes:
// Vector3 has no native wasm valtype, so it lives as a v128 with the low 12 bytes
// populated. The frame address is pushed twice: v128.load64_zero fills lanes 0-1
// (zeroing the rest) and v128.load32_lane fills lane 2 from bytes 8-11.
//
void CodeGen::genLoadLclTypeSimd12(GenTreeLclVarCommon* tree)
{
bool fpBased;
int frameOffset = m_compiler->lvaFrameAddress(tree->GetLclNum(), &fpBased) + (int)tree->GetLclOffs();
noway_assert(frameOffset >= 0); // WASM address modes are unsigned.
assert(fpBased);
unsigned fpIndex = GetFramePointerRegIndex();
emitter* emit = GetEmitter();

emit->emitIns_I(INS_local_get, EA_PTRSIZE, fpIndex);
emit->emitIns_I(INS_local_get, EA_PTRSIZE, fpIndex);
emit->emitIns_I(INS_v128_load64_zero, EA_8BYTE, frameOffset);
emit->emitIns_MemargLane(INS_v128_load32_lane, EA_4BYTE, frameOffset + 8, 2);
}

//------------------------------------------------------------------------
// genLoadIndTypeSimd12: Load a TYP_SIMD12 (i.e. Vector3) value through an indirection.
//
// Arguments:
// tree - the GT_IND node
//
// Notes:
// The address is left on the value stack by prior codegen and is multiply-used, so the
// trailing v128.load32_lane can re-push it for the upper 4 bytes.
//
void CodeGen::genLoadIndTypeSimd12(GenTreeIndir* tree)
{
emitter* emit = GetEmitter();

emit->emitIns_I(INS_local_get, EA_PTRSIZE, WasmRegToIndex(GetMultiUseOperandReg(tree->Addr())));
emit->emitIns_I(INS_v128_load64_zero, EA_8BYTE, 0);
emit->emitIns_MemargLane(INS_v128_load32_lane, EA_4BYTE, 8, 2);
}

//------------------------------------------------------------------------
// genStoreIndTypeSimd12: Store a TYP_SIMD12 (i.e. Vector3) value through an indirection.
//
// Arguments:
// tree - the GT_STOREIND node
//
// Notes:
// On entry the value stack holds [addr, value]. The value is teed into an internal v128
// local so it survives the low-8 store; the address is then re-materialized to store the
// upper 4 bytes via a lane store - re-emitting the frame pointer for a LCL_ADDR, or
// re-pushing the multiply-used address register otherwise.
//
void CodeGen::genStoreIndTypeSimd12(GenTreeStoreInd* tree)
{
emitter* emit = GetEmitter();
GenTree* addr = tree->Addr();

InternalRegs* regs = internalRegisters.GetAll(tree);
assert(regs->Count() == 1);
regNumber valReg = regs->Extract();

emit->emitIns_I(INS_local_tee, EA_16BYTE, WasmRegToIndex(valReg)); // [addr, value]
emit->emitIns_MemargLane(INS_v128_store64_lane, EA_8BYTE, 0, 0); // []

if (addr->OperIs(GT_LCL_ADDR))
{
bool fpBased;
int frameOffset = m_compiler->lvaFrameAddress(addr->AsLclVarCommon()->GetLclNum(), &fpBased) +
(int)addr->AsLclVarCommon()->GetLclOffs();
noway_assert(frameOffset >= 0); // WASM address modes are unsigned.
assert(fpBased);
emit->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex()); // [fp]
emit->emitIns_I(INS_local_get, EA_16BYTE, WasmRegToIndex(valReg)); // [fp, value]
emit->emitIns_MemargLane(INS_v128_store32_lane, EA_4BYTE, frameOffset + 8, 2); // []
}
else
{
emit->emitIns_I(INS_local_get, EA_PTRSIZE, WasmRegToIndex(GetMultiUseOperandReg(addr))); // [addr]
emit->emitIns_I(INS_local_get, EA_16BYTE, WasmRegToIndex(valReg)); // [addr, value]
emit->emitIns_MemargLane(INS_v128_store32_lane, EA_4BYTE, 8, 2); // []
}
}

//------------------------------------------------------------------------
// genCodeForIndir: Produce code for a GT_IND node.
//
Expand All@@ -2705,8 +2807,7 @@ void CodeGen::genCodeForIndir(GenTreeIndir* tree)
{
assert(tree->OperIs(GT_IND));

var_types type = tree->TypeGet();
instruction ins = ins_Load(type);
var_types type = tree->TypeGet();

genConsumeAddress(tree->Addr());

Expand All@@ -2718,7 +2819,14 @@ void CodeGen::genCodeForIndir(GenTreeIndir* tree)

// TODO-WASM: Memory barriers

GetEmitter()->emitIns_I(ins, emitActualTypeSize(type), 0);
if (type == TYP_SIMD12)
{
genLoadIndTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(ins_Load(type), emitActualTypeSize(type), 0);
}

WasmProduceReg(tree);
}
Expand DownExpand Up@@ -2762,11 +2870,22 @@ void CodeGen::genCodeForStoreInd(GenTreeStoreInd* tree)
// module. Bail until SIMD16 store is properly supported.
NYI_WASM_SIMD("SIMD16 store indirect");
}
instruction ins = ins_Store(type);

// TODO-WASM: Memory barriers

GetEmitter()->emitIns_I(ins, emitActualTypeSize(type), 0);
if (type == TYP_SIMD8)
{
// stack: [addr, value] -> store the low 8 bytes.
GetEmitter()->emitIns_MemargLane(INS_v128_store64_lane, EA_8BYTE, 0, 0);
}
else if (type == TYP_SIMD12)
{
genStoreIndTypeSimd12(tree);
}
Comment thread
tannergooding marked this conversation as resolved.
else
{
GetEmitter()->emitIns_I(ins_Store(type), emitActualTypeSize(type), 0);
}
}

genUpdateLife(tree);
Expand Down
4 changes: 4 additions & 0 deletions src/coreclr/jit/instr.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -2070,6 +2070,10 @@ instruction CodeGenInterface::ins_Load(var_types srcType, bool aligned /*=false*
case TYP_DOUBLE:
return INS_f64_load;
#if defined(FEATURE_SIMD)
case TYP_SIMD8:
// SIMD8 (Vector2) lives as a v128 with the low 8 bytes populated. SIMD12 (Vector3) is
// handled at the callers since it needs a trailing lane load for the upper 4 bytes.
return INS_v128_load64_zero;
case TYP_SIMD16:
return INS_v128_load;
#endif
Expand Down
14 changes: 10 additions & 4 deletions src/coreclr/jit/lowerwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -164,10 +164,14 @@ GenTree* Lowering::LowerStoreLoc(GenTreeLclVarCommon* storeLoc)
//
GenTree* Lowering::LowerStoreIndir(GenTreeStoreInd* node)
{
if ((node->gtFlags & GTF_IND_NONFAULTING) == 0)
if (((node->gtFlags & GTF_IND_NONFAULTING) == 0) ||
(node->TypeIs(TYP_SIMD12) && !node->Addr()->OperIs(GT_LCL_ADDR)))
{
// We need to be able to null check the address, and that requires multiple uses of the address operand.
SetMultiplyUsed(node->Addr() DEBUGARG("LowerStoreIndir faulting Addr"));
// SIMD12 stores also re-materialize the address for the trailing lane store, so force it there as well -
// unless the address is a re-materializable LCL_ADDR (the local-to-stack store rewrite), which codegen
// re-emits directly.
SetMultiplyUsed(node->Addr() DEBUGARG("LowerStoreIndir Addr (null check or simd12 lane store)"));
}

ContainCheckStoreIndir(node);
Expand DownExpand Up@@ -459,9 +463,11 @@ void Lowering::ContainCheckIndir(GenTreeIndir* indirNode)
return;
}

if (indirNode->OperIs(GT_IND) && ((indirNode->gtFlags & GTF_IND_NONFAULTING) == 0))
if (indirNode->OperIs(GT_IND) &&
(((indirNode->gtFlags & GTF_IND_NONFAULTING) == 0) || indirNode->TypeIs(TYP_SIMD12)))
{
SetMultiplyUsed(indirNode->Addr() DEBUGARG("ContainCheckIndir faulting load Addr"));
// SIMD12 loads re-materialize the address for the trailing lane load, so force it there regardless.
SetMultiplyUsed(indirNode->Addr() DEBUGARG("ContainCheckIndir load Addr (null check or simd12 lane load)"));
}

// TODO-WASM-CQ: contain suitable LEAs here. Take note of the fact that for this to be correct we must prove the
Expand Down
18 changes: 18 additions & 0 deletions src/coreclr/jit/regallocwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -662,6 +662,14 @@ void WasmRegAlloc::CollectReferencesForIndir(GenTreeIndir* node)
{
GenTree* const addr = node->Addr();
ConsumeTemporaryRegForOperand(addr DEBUGARG("indirection address"));

if (node->OperIs(GT_STOREIND) && node->TypeIs(TYP_SIMD12))
{
// The SIMD12 store stashes the v128 value so it can re-push it for the trailing lane store.
regNumber internalReg = RequestInternalRegister(node, TYP_SIMD16);
regNumber releasedReg = ReleaseTemporaryRegister(WasmRegToType(internalReg));
assert(releasedReg == internalReg);
}
}

//------------------------------------------------------------------------
Expand DownExpand Up@@ -800,6 +808,16 @@ void WasmRegAlloc::RewriteLocalStackStore(GenTreeLclVarCommon* lclNode)
LIR::ReadOnlyRange storeRange(store, store);
m_compiler->GetLowering()->LowerRange(m_currentBlock, storeRange);

if (store->OperIs(GT_STOREIND) && store->TypeIs(TYP_SIMD12))
{
// genStoreIndTypeSimd12 tees the value into a v128 temporary to split the store into an 8-byte and a
// 4-byte lane store. The main collection walk does not revisit this freshly-introduced node, so request
// that internal register here. The re-materializable LCL_ADDR address needs no temporary.
regNumber internalReg = RequestInternalRegister(store, TYP_SIMD16);
regNumber releasedReg = ReleaseTemporaryRegister(WasmRegToType(internalReg));
assert(releasedReg == internalReg);
}

// FIXME-WASM: Should we be doing this here?
// CollectReferencesForNode(store);
}
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Strip utm_, fbclid, gclid, etc. from all links on page (function() { var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content', 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid', 'ref', 'ref_src', 'source', 'medium', 'campaign']; function cleanUrl(url) { try { var u = new URL(url, window.location.origin); var changed = false; trackingParams.forEach(function(p) { if (u.searchParams.has(p)) { u.searchParams.delete(p); changed = true; } }); return changed ? u.toString() : url; } catch (e) { return url; } } function cleanLinks() { document.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } cleanLinks(); var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1) { if (node.tagName === 'A') cleanLinks(); node.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
137 changes: 128 additions & 9 deletions src/coreclr/jit/codegenwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -2597,8 +2597,15 @@ void CodeGen::genCodeForLclFld(GenTreeLclFld* tree)
NYI_WASM_SIMD("SIMD16 local field load");
}

GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(tree), tree->GetLclNum(), tree->GetLclOffs());
if (type == TYP_SIMD12)
{
genLoadLclTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(tree), tree->GetLclNum(), tree->GetLclOffs());
}
WasmProduceReg(tree);
}

Expand All@@ -2621,8 +2628,15 @@ void CodeGen::genCodeForLclVar(GenTreeLclVar* tree)
{
var_types type = varDsc->GetRegisterType(tree);

GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(type), tree->GetLclNum(), 0);
if (type == TYP_SIMD12)
{
genLoadLclTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(type), tree->GetLclNum(), 0);
}
WasmProduceReg(tree);
}
else
Expand DownExpand Up@@ -2695,6 +2709,94 @@ void CodeGen::genCodeForFrameSize(GenTree* tree)
WasmProduceReg(tree);
}

//------------------------------------------------------------------------
// genLoadLclTypeSimd12: Load a TYP_SIMD12 (i.e. Vector3) local into a v128.
//
// Arguments:
// tree - the GT_LCL_FLD or GT_LCL_VAR node
//
// Notes:
// Vector3 has no native wasm valtype, so it lives as a v128 with the low 12 bytes
// populated. The frame address is pushed twice: v128.load64_zero fills lanes 0-1
// (zeroing the rest) and v128.load32_lane fills lane 2 from bytes 8-11.
//
void CodeGen::genLoadLclTypeSimd12(GenTreeLclVarCommon* tree)
{
bool fpBased;
int frameOffset = m_compiler->lvaFrameAddress(tree->GetLclNum(), &fpBased) + (int)tree->GetLclOffs();
noway_assert(frameOffset >= 0); // WASM address modes are unsigned.
assert(fpBased);
unsigned fpIndex = GetFramePointerRegIndex();
emitter* emit = GetEmitter();

emit->emitIns_I(INS_local_get, EA_PTRSIZE, fpIndex);
emit->emitIns_I(INS_local_get, EA_PTRSIZE, fpIndex);
emit->emitIns_I(INS_v128_load64_zero, EA_8BYTE, frameOffset);
emit->emitIns_MemargLane(INS_v128_load32_lane, EA_4BYTE, frameOffset + 8, 2);
}

//------------------------------------------------------------------------
// genLoadIndTypeSimd12: Load a TYP_SIMD12 (i.e. Vector3) value through an indirection.
//
// Arguments:
// tree - the GT_IND node
//
// Notes:
// The address is left on the value stack by prior codegen and is multiply-used, so the
// trailing v128.load32_lane can re-push it for the upper 4 bytes.
//
void CodeGen::genLoadIndTypeSimd12(GenTreeIndir* tree)
{
emitter* emit = GetEmitter();

emit->emitIns_I(INS_local_get, EA_PTRSIZE, WasmRegToIndex(GetMultiUseOperandReg(tree->Addr())));
emit->emitIns_I(INS_v128_load64_zero, EA_8BYTE, 0);
emit->emitIns_MemargLane(INS_v128_load32_lane, EA_4BYTE, 8, 2);
}

//------------------------------------------------------------------------
// genStoreIndTypeSimd12: Store a TYP_SIMD12 (i.e. Vector3) value through an indirection.
//
// Arguments:
// tree - the GT_STOREIND node
//
// Notes:
// On entry the value stack holds [addr, value]. The value is teed into an internal v128
// local so it survives the low-8 store; the address is then re-materialized to store the
// upper 4 bytes via a lane store - re-emitting the frame pointer for a LCL_ADDR, or
// re-pushing the multiply-used address register otherwise.
//
void CodeGen::genStoreIndTypeSimd12(GenTreeStoreInd* tree)
{
emitter* emit = GetEmitter();
GenTree* addr = tree->Addr();

InternalRegs* regs = internalRegisters.GetAll(tree);
assert(regs->Count() == 1);
regNumber valReg = regs->Extract();

emit->emitIns_I(INS_local_tee, EA_16BYTE, WasmRegToIndex(valReg)); // [addr, value]
emit->emitIns_MemargLane(INS_v128_store64_lane, EA_8BYTE, 0, 0); // []

if (addr->OperIs(GT_LCL_ADDR))
{
bool fpBased;
int frameOffset = m_compiler->lvaFrameAddress(addr->AsLclVarCommon()->GetLclNum(), &fpBased) +
(int)addr->AsLclVarCommon()->GetLclOffs();
noway_assert(frameOffset >= 0); // WASM address modes are unsigned.
assert(fpBased);
emit->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex()); // [fp]
emit->emitIns_I(INS_local_get, EA_16BYTE, WasmRegToIndex(valReg)); // [fp, value]
emit->emitIns_MemargLane(INS_v128_store32_lane, EA_4BYTE, frameOffset + 8, 2); // []
}
else
{
emit->emitIns_I(INS_local_get, EA_PTRSIZE, WasmRegToIndex(GetMultiUseOperandReg(addr))); // [addr]
emit->emitIns_I(INS_local_get, EA_16BYTE, WasmRegToIndex(valReg)); // [addr, value]
emit->emitIns_MemargLane(INS_v128_store32_lane, EA_4BYTE, 8, 2); // []
}
}

//------------------------------------------------------------------------
// genCodeForIndir: Produce code for a GT_IND node.
//
Expand All@@ -2705,8 +2807,7 @@ void CodeGen::genCodeForIndir(GenTreeIndir* tree)
{
assert(tree->OperIs(GT_IND));

var_types type = tree->TypeGet();
instruction ins = ins_Load(type);
var_types type = tree->TypeGet();

genConsumeAddress(tree->Addr());

Expand All@@ -2718,7 +2819,14 @@ void CodeGen::genCodeForIndir(GenTreeIndir* tree)

// TODO-WASM: Memory barriers

GetEmitter()->emitIns_I(ins, emitActualTypeSize(type), 0);
if (type == TYP_SIMD12)
{
genLoadIndTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(ins_Load(type), emitActualTypeSize(type), 0);
}

WasmProduceReg(tree);
}
Expand DownExpand Up@@ -2762,11 +2870,22 @@ void CodeGen::genCodeForStoreInd(GenTreeStoreInd* tree)
// module. Bail until SIMD16 store is properly supported.
NYI_WASM_SIMD("SIMD16 store indirect");
}
instruction ins = ins_Store(type);

// TODO-WASM: Memory barriers

GetEmitter()->emitIns_I(ins, emitActualTypeSize(type), 0);
if (type == TYP_SIMD8)
{
// stack: [addr, value] -> store the low 8 bytes.
GetEmitter()->emitIns_MemargLane(INS_v128_store64_lane, EA_8BYTE, 0, 0);
}
else if (type == TYP_SIMD12)
{
genStoreIndTypeSimd12(tree);
}
Comment thread
tannergooding marked this conversation as resolved.
else
{
GetEmitter()->emitIns_I(ins_Store(type), emitActualTypeSize(type), 0);
}
}

genUpdateLife(tree);
Expand Down
4 changes: 4 additions & 0 deletions src/coreclr/jit/instr.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -2070,6 +2070,10 @@ instruction CodeGenInterface::ins_Load(var_types srcType, bool aligned /*=false*
case TYP_DOUBLE:
return INS_f64_load;
#if defined(FEATURE_SIMD)
case TYP_SIMD8:
// SIMD8 (Vector2) lives as a v128 with the low 8 bytes populated. SIMD12 (Vector3) is
// handled at the callers since it needs a trailing lane load for the upper 4 bytes.
return INS_v128_load64_zero;
case TYP_SIMD16:
return INS_v128_load;
#endif
Expand Down
14 changes: 10 additions & 4 deletions src/coreclr/jit/lowerwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -164,10 +164,14 @@ GenTree* Lowering::LowerStoreLoc(GenTreeLclVarCommon* storeLoc)
//
GenTree* Lowering::LowerStoreIndir(GenTreeStoreInd* node)
{
if ((node->gtFlags & GTF_IND_NONFAULTING) == 0)
if (((node->gtFlags & GTF_IND_NONFAULTING) == 0) ||
(node->TypeIs(TYP_SIMD12) && !node->Addr()->OperIs(GT_LCL_ADDR)))
{
// We need to be able to null check the address, and that requires multiple uses of the address operand.
SetMultiplyUsed(node->Addr() DEBUGARG("LowerStoreIndir faulting Addr"));
// SIMD12 stores also re-materialize the address for the trailing lane store, so force it there as well -
// unless the address is a re-materializable LCL_ADDR (the local-to-stack store rewrite), which codegen
// re-emits directly.
SetMultiplyUsed(node->Addr() DEBUGARG("LowerStoreIndir Addr (null check or simd12 lane store)"));
}

ContainCheckStoreIndir(node);
Expand DownExpand Up@@ -459,9 +463,11 @@ void Lowering::ContainCheckIndir(GenTreeIndir* indirNode)
return;
}

if (indirNode->OperIs(GT_IND) && ((indirNode->gtFlags & GTF_IND_NONFAULTING) == 0))
if (indirNode->OperIs(GT_IND) &&
(((indirNode->gtFlags & GTF_IND_NONFAULTING) == 0) || indirNode->TypeIs(TYP_SIMD12)))
{
SetMultiplyUsed(indirNode->Addr() DEBUGARG("ContainCheckIndir faulting load Addr"));
// SIMD12 loads re-materialize the address for the trailing lane load, so force it there regardless.
SetMultiplyUsed(indirNode->Addr() DEBUGARG("ContainCheckIndir load Addr (null check or simd12 lane load)"));
}

// TODO-WASM-CQ: contain suitable LEAs here. Take note of the fact that for this to be correct we must prove the
Expand Down
18 changes: 18 additions & 0 deletions src/coreclr/jit/regallocwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -662,6 +662,14 @@ void WasmRegAlloc::CollectReferencesForIndir(GenTreeIndir* node)
{
GenTree* const addr = node->Addr();
ConsumeTemporaryRegForOperand(addr DEBUGARG("indirection address"));

if (node->OperIs(GT_STOREIND) && node->TypeIs(TYP_SIMD12))
{
// The SIMD12 store stashes the v128 value so it can re-push it for the trailing lane store.
regNumber internalReg = RequestInternalRegister(node, TYP_SIMD16);
regNumber releasedReg = ReleaseTemporaryRegister(WasmRegToType(internalReg));
assert(releasedReg == internalReg);
}
}

//------------------------------------------------------------------------
Expand DownExpand Up@@ -800,6 +808,16 @@ void WasmRegAlloc::RewriteLocalStackStore(GenTreeLclVarCommon* lclNode)
LIR::ReadOnlyRange storeRange(store, store);
m_compiler->GetLowering()->LowerRange(m_currentBlock, storeRange);

if (store->OperIs(GT_STOREIND) && store->TypeIs(TYP_SIMD12))
{
// genStoreIndTypeSimd12 tees the value into a v128 temporary to split the store into an 8-byte and a
// 4-byte lane store. The main collection walk does not revisit this freshly-introduced node, so request
// that internal register here. The re-materializable LCL_ADDR address needs no temporary.
regNumber internalReg = RequestInternalRegister(store, TYP_SIMD16);
regNumber releasedReg = ReleaseTemporaryRegister(WasmRegToType(internalReg));
assert(releasedReg == internalReg);
}

// FIXME-WASM: Should we be doing this here?
// CollectReferencesForNode(store);
}
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Auto-enable theater mode on YouTube (function() { function tryTheater() { var btn = document.querySelector('button[aria-label="Theater mode"], ytd-player #player button[title="Theater mode"]'); if (btn && !btn.classList.contains('activated')) { btn.click(); } } // Try immediately tryTheater(); // Try after navigation (SPA) var lastUrl = location.href; setInterval(function() { if (location.href !== lastUrl) { lastUrl = location.href; setTimeout(tryTheater, 500); } }, 1000); // Also try on player load var observer = new MutationObserver(tryTheater); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
137 changes: 128 additions & 9 deletions src/coreclr/jit/codegenwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -2597,8 +2597,15 @@ void CodeGen::genCodeForLclFld(GenTreeLclFld* tree)
NYI_WASM_SIMD("SIMD16 local field load");
}

GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(tree), tree->GetLclNum(), tree->GetLclOffs());
if (type == TYP_SIMD12)
{
genLoadLclTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(tree), tree->GetLclNum(), tree->GetLclOffs());
}
WasmProduceReg(tree);
}

Expand All@@ -2621,8 +2628,15 @@ void CodeGen::genCodeForLclVar(GenTreeLclVar* tree)
{
var_types type = varDsc->GetRegisterType(tree);

GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(type), tree->GetLclNum(), 0);
if (type == TYP_SIMD12)
{
genLoadLclTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(type), tree->GetLclNum(), 0);
}
WasmProduceReg(tree);
}
else
Expand DownExpand Up@@ -2695,6 +2709,94 @@ void CodeGen::genCodeForFrameSize(GenTree* tree)
WasmProduceReg(tree);
}

//------------------------------------------------------------------------
// genLoadLclTypeSimd12: Load a TYP_SIMD12 (i.e. Vector3) local into a v128.
//
// Arguments:
// tree - the GT_LCL_FLD or GT_LCL_VAR node
//
// Notes:
// Vector3 has no native wasm valtype, so it lives as a v128 with the low 12 bytes
// populated. The frame address is pushed twice: v128.load64_zero fills lanes 0-1
// (zeroing the rest) and v128.load32_lane fills lane 2 from bytes 8-11.
//
void CodeGen::genLoadLclTypeSimd12(GenTreeLclVarCommon* tree)
{
bool fpBased;
int frameOffset = m_compiler->lvaFrameAddress(tree->GetLclNum(), &fpBased) + (int)tree->GetLclOffs();
noway_assert(frameOffset >= 0); // WASM address modes are unsigned.
assert(fpBased);
unsigned fpIndex = GetFramePointerRegIndex();
emitter* emit = GetEmitter();

emit->emitIns_I(INS_local_get, EA_PTRSIZE, fpIndex);
emit->emitIns_I(INS_local_get, EA_PTRSIZE, fpIndex);
emit->emitIns_I(INS_v128_load64_zero, EA_8BYTE, frameOffset);
emit->emitIns_MemargLane(INS_v128_load32_lane, EA_4BYTE, frameOffset + 8, 2);
}

//------------------------------------------------------------------------
// genLoadIndTypeSimd12: Load a TYP_SIMD12 (i.e. Vector3) value through an indirection.
//
// Arguments:
// tree - the GT_IND node
//
// Notes:
// The address is left on the value stack by prior codegen and is multiply-used, so the
// trailing v128.load32_lane can re-push it for the upper 4 bytes.
//
void CodeGen::genLoadIndTypeSimd12(GenTreeIndir* tree)
{
emitter* emit = GetEmitter();

emit->emitIns_I(INS_local_get, EA_PTRSIZE, WasmRegToIndex(GetMultiUseOperandReg(tree->Addr())));
emit->emitIns_I(INS_v128_load64_zero, EA_8BYTE, 0);
emit->emitIns_MemargLane(INS_v128_load32_lane, EA_4BYTE, 8, 2);
}

//------------------------------------------------------------------------
// genStoreIndTypeSimd12: Store a TYP_SIMD12 (i.e. Vector3) value through an indirection.
//
// Arguments:
// tree - the GT_STOREIND node
//
// Notes:
// On entry the value stack holds [addr, value]. The value is teed into an internal v128
// local so it survives the low-8 store; the address is then re-materialized to store the
// upper 4 bytes via a lane store - re-emitting the frame pointer for a LCL_ADDR, or
// re-pushing the multiply-used address register otherwise.
//
void CodeGen::genStoreIndTypeSimd12(GenTreeStoreInd* tree)
{
emitter* emit = GetEmitter();
GenTree* addr = tree->Addr();

InternalRegs* regs = internalRegisters.GetAll(tree);
assert(regs->Count() == 1);
regNumber valReg = regs->Extract();

emit->emitIns_I(INS_local_tee, EA_16BYTE, WasmRegToIndex(valReg)); // [addr, value]
emit->emitIns_MemargLane(INS_v128_store64_lane, EA_8BYTE, 0, 0); // []

if (addr->OperIs(GT_LCL_ADDR))
{
bool fpBased;
int frameOffset = m_compiler->lvaFrameAddress(addr->AsLclVarCommon()->GetLclNum(), &fpBased) +
(int)addr->AsLclVarCommon()->GetLclOffs();
noway_assert(frameOffset >= 0); // WASM address modes are unsigned.
assert(fpBased);
emit->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex()); // [fp]
emit->emitIns_I(INS_local_get, EA_16BYTE, WasmRegToIndex(valReg)); // [fp, value]
emit->emitIns_MemargLane(INS_v128_store32_lane, EA_4BYTE, frameOffset + 8, 2); // []
}
else
{
emit->emitIns_I(INS_local_get, EA_PTRSIZE, WasmRegToIndex(GetMultiUseOperandReg(addr))); // [addr]
emit->emitIns_I(INS_local_get, EA_16BYTE, WasmRegToIndex(valReg)); // [addr, value]
emit->emitIns_MemargLane(INS_v128_store32_lane, EA_4BYTE, 8, 2); // []
}
}

//------------------------------------------------------------------------
// genCodeForIndir: Produce code for a GT_IND node.
//
Expand All@@ -2705,8 +2807,7 @@ void CodeGen::genCodeForIndir(GenTreeIndir* tree)
{
assert(tree->OperIs(GT_IND));

var_types type = tree->TypeGet();
instruction ins = ins_Load(type);
var_types type = tree->TypeGet();

genConsumeAddress(tree->Addr());

Expand All@@ -2718,7 +2819,14 @@ void CodeGen::genCodeForIndir(GenTreeIndir* tree)

// TODO-WASM: Memory barriers

GetEmitter()->emitIns_I(ins, emitActualTypeSize(type), 0);
if (type == TYP_SIMD12)
{
genLoadIndTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(ins_Load(type), emitActualTypeSize(type), 0);
}

WasmProduceReg(tree);
}
Expand DownExpand Up@@ -2762,11 +2870,22 @@ void CodeGen::genCodeForStoreInd(GenTreeStoreInd* tree)
// module. Bail until SIMD16 store is properly supported.
NYI_WASM_SIMD("SIMD16 store indirect");
}
instruction ins = ins_Store(type);

// TODO-WASM: Memory barriers

GetEmitter()->emitIns_I(ins, emitActualTypeSize(type), 0);
if (type == TYP_SIMD8)
{
// stack: [addr, value] -> store the low 8 bytes.
GetEmitter()->emitIns_MemargLane(INS_v128_store64_lane, EA_8BYTE, 0, 0);
}
else if (type == TYP_SIMD12)
{
genStoreIndTypeSimd12(tree);
}
Comment thread
tannergooding marked this conversation as resolved.
else
{
GetEmitter()->emitIns_I(ins_Store(type), emitActualTypeSize(type), 0);
}
}

genUpdateLife(tree);
Expand Down
4 changes: 4 additions & 0 deletions src/coreclr/jit/instr.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -2070,6 +2070,10 @@ instruction CodeGenInterface::ins_Load(var_types srcType, bool aligned /*=false*
case TYP_DOUBLE:
return INS_f64_load;
#if defined(FEATURE_SIMD)
case TYP_SIMD8:
// SIMD8 (Vector2) lives as a v128 with the low 8 bytes populated. SIMD12 (Vector3) is
// handled at the callers since it needs a trailing lane load for the upper 4 bytes.
return INS_v128_load64_zero;
case TYP_SIMD16:
return INS_v128_load;
#endif
Expand Down
14 changes: 10 additions & 4 deletions src/coreclr/jit/lowerwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -164,10 +164,14 @@ GenTree* Lowering::LowerStoreLoc(GenTreeLclVarCommon* storeLoc)
//
GenTree* Lowering::LowerStoreIndir(GenTreeStoreInd* node)
{
if ((node->gtFlags & GTF_IND_NONFAULTING) == 0)
if (((node->gtFlags & GTF_IND_NONFAULTING) == 0) ||
(node->TypeIs(TYP_SIMD12) && !node->Addr()->OperIs(GT_LCL_ADDR)))
{
// We need to be able to null check the address, and that requires multiple uses of the address operand.
SetMultiplyUsed(node->Addr() DEBUGARG("LowerStoreIndir faulting Addr"));
// SIMD12 stores also re-materialize the address for the trailing lane store, so force it there as well -
// unless the address is a re-materializable LCL_ADDR (the local-to-stack store rewrite), which codegen
// re-emits directly.
SetMultiplyUsed(node->Addr() DEBUGARG("LowerStoreIndir Addr (null check or simd12 lane store)"));
}

ContainCheckStoreIndir(node);
Expand DownExpand Up@@ -459,9 +463,11 @@ void Lowering::ContainCheckIndir(GenTreeIndir* indirNode)
return;
}

if (indirNode->OperIs(GT_IND) && ((indirNode->gtFlags & GTF_IND_NONFAULTING) == 0))
if (indirNode->OperIs(GT_IND) &&
(((indirNode->gtFlags & GTF_IND_NONFAULTING) == 0) || indirNode->TypeIs(TYP_SIMD12)))
{
SetMultiplyUsed(indirNode->Addr() DEBUGARG("ContainCheckIndir faulting load Addr"));
// SIMD12 loads re-materialize the address for the trailing lane load, so force it there regardless.
SetMultiplyUsed(indirNode->Addr() DEBUGARG("ContainCheckIndir load Addr (null check or simd12 lane load)"));
}

// TODO-WASM-CQ: contain suitable LEAs here. Take note of the fact that for this to be correct we must prove the
Expand Down
18 changes: 18 additions & 0 deletions src/coreclr/jit/regallocwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -662,6 +662,14 @@ void WasmRegAlloc::CollectReferencesForIndir(GenTreeIndir* node)
{
GenTree* const addr = node->Addr();
ConsumeTemporaryRegForOperand(addr DEBUGARG("indirection address"));

if (node->OperIs(GT_STOREIND) && node->TypeIs(TYP_SIMD12))
{
// The SIMD12 store stashes the v128 value so it can re-push it for the trailing lane store.
regNumber internalReg = RequestInternalRegister(node, TYP_SIMD16);
regNumber releasedReg = ReleaseTemporaryRegister(WasmRegToType(internalReg));
assert(releasedReg == internalReg);
}
}

//------------------------------------------------------------------------
Expand DownExpand Up@@ -800,6 +808,16 @@ void WasmRegAlloc::RewriteLocalStackStore(GenTreeLclVarCommon* lclNode)
LIR::ReadOnlyRange storeRange(store, store);
m_compiler->GetLowering()->LowerRange(m_currentBlock, storeRange);

if (store->OperIs(GT_STOREIND) && store->TypeIs(TYP_SIMD12))
{
// genStoreIndTypeSimd12 tees the value into a v128 temporary to split the store into an 8-byte and a
// 4-byte lane store. The main collection walk does not revisit this freshly-introduced node, so request
// that internal register here. The re-materializable LCL_ADDR address needs no temporary.
regNumber internalReg = RequestInternalRegister(store, TYP_SIMD16);
regNumber releasedReg = ReleaseTemporaryRegister(WasmRegToType(internalReg));
assert(releasedReg == internalReg);
}

// FIXME-WASM: Should we be doing this here?
// CollectReferencesForNode(store);
}
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Remove or un-stick sticky/fixed headers that block content (function() { function unstick() { document.querySelectorAll('header, nav, [role="banner"], .header, .navbar, .sticky, .fixed-top, [style*="position: fixed"], [style*="position:sticky"]').forEach(function(el) { if (el.style.position === 'fixed' || el.style.position === 'sticky' || getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') { el.style.position = 'static'; el.style.top = 'auto'; el.style.zIndex = 'auto'; } }); } unstick(); var observer = new MutationObserver(unstick); observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] }); })(); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
137 changes: 128 additions & 9 deletions src/coreclr/jit/codegenwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -2597,8 +2597,15 @@ void CodeGen::genCodeForLclFld(GenTreeLclFld* tree)
NYI_WASM_SIMD("SIMD16 local field load");
}

GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(tree), tree->GetLclNum(), tree->GetLclOffs());
if (type == TYP_SIMD12)
{
genLoadLclTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(tree), tree->GetLclNum(), tree->GetLclOffs());
}
WasmProduceReg(tree);
}

Expand All@@ -2621,8 +2628,15 @@ void CodeGen::genCodeForLclVar(GenTreeLclVar* tree)
{
var_types type = varDsc->GetRegisterType(tree);

GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(type), tree->GetLclNum(), 0);
if (type == TYP_SIMD12)
{
genLoadLclTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(type), tree->GetLclNum(), 0);
}
WasmProduceReg(tree);
}
else
Expand DownExpand Up@@ -2695,6 +2709,94 @@ void CodeGen::genCodeForFrameSize(GenTree* tree)
WasmProduceReg(tree);
}

//------------------------------------------------------------------------
// genLoadLclTypeSimd12: Load a TYP_SIMD12 (i.e. Vector3) local into a v128.
//
// Arguments:
// tree - the GT_LCL_FLD or GT_LCL_VAR node
//
// Notes:
// Vector3 has no native wasm valtype, so it lives as a v128 with the low 12 bytes
// populated. The frame address is pushed twice: v128.load64_zero fills lanes 0-1
// (zeroing the rest) and v128.load32_lane fills lane 2 from bytes 8-11.
//
void CodeGen::genLoadLclTypeSimd12(GenTreeLclVarCommon* tree)
{
bool fpBased;
int frameOffset = m_compiler->lvaFrameAddress(tree->GetLclNum(), &fpBased) + (int)tree->GetLclOffs();
noway_assert(frameOffset >= 0); // WASM address modes are unsigned.
assert(fpBased);
unsigned fpIndex = GetFramePointerRegIndex();
emitter* emit = GetEmitter();

emit->emitIns_I(INS_local_get, EA_PTRSIZE, fpIndex);
emit->emitIns_I(INS_local_get, EA_PTRSIZE, fpIndex);
emit->emitIns_I(INS_v128_load64_zero, EA_8BYTE, frameOffset);
emit->emitIns_MemargLane(INS_v128_load32_lane, EA_4BYTE, frameOffset + 8, 2);
}

//------------------------------------------------------------------------
// genLoadIndTypeSimd12: Load a TYP_SIMD12 (i.e. Vector3) value through an indirection.
//
// Arguments:
// tree - the GT_IND node
//
// Notes:
// The address is left on the value stack by prior codegen and is multiply-used, so the
// trailing v128.load32_lane can re-push it for the upper 4 bytes.
//
void CodeGen::genLoadIndTypeSimd12(GenTreeIndir* tree)
{
emitter* emit = GetEmitter();

emit->emitIns_I(INS_local_get, EA_PTRSIZE, WasmRegToIndex(GetMultiUseOperandReg(tree->Addr())));
emit->emitIns_I(INS_v128_load64_zero, EA_8BYTE, 0);
emit->emitIns_MemargLane(INS_v128_load32_lane, EA_4BYTE, 8, 2);
}

//------------------------------------------------------------------------
// genStoreIndTypeSimd12: Store a TYP_SIMD12 (i.e. Vector3) value through an indirection.
//
// Arguments:
// tree - the GT_STOREIND node
//
// Notes:
// On entry the value stack holds [addr, value]. The value is teed into an internal v128
// local so it survives the low-8 store; the address is then re-materialized to store the
// upper 4 bytes via a lane store - re-emitting the frame pointer for a LCL_ADDR, or
// re-pushing the multiply-used address register otherwise.
//
void CodeGen::genStoreIndTypeSimd12(GenTreeStoreInd* tree)
{
emitter* emit = GetEmitter();
GenTree* addr = tree->Addr();

InternalRegs* regs = internalRegisters.GetAll(tree);
assert(regs->Count() == 1);
regNumber valReg = regs->Extract();

emit->emitIns_I(INS_local_tee, EA_16BYTE, WasmRegToIndex(valReg)); // [addr, value]
emit->emitIns_MemargLane(INS_v128_store64_lane, EA_8BYTE, 0, 0); // []

if (addr->OperIs(GT_LCL_ADDR))
{
bool fpBased;
int frameOffset = m_compiler->lvaFrameAddress(addr->AsLclVarCommon()->GetLclNum(), &fpBased) +
(int)addr->AsLclVarCommon()->GetLclOffs();
noway_assert(frameOffset >= 0); // WASM address modes are unsigned.
assert(fpBased);
emit->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex()); // [fp]
emit->emitIns_I(INS_local_get, EA_16BYTE, WasmRegToIndex(valReg)); // [fp, value]
emit->emitIns_MemargLane(INS_v128_store32_lane, EA_4BYTE, frameOffset + 8, 2); // []
}
else
{
emit->emitIns_I(INS_local_get, EA_PTRSIZE, WasmRegToIndex(GetMultiUseOperandReg(addr))); // [addr]
emit->emitIns_I(INS_local_get, EA_16BYTE, WasmRegToIndex(valReg)); // [addr, value]
emit->emitIns_MemargLane(INS_v128_store32_lane, EA_4BYTE, 8, 2); // []
}
}

//------------------------------------------------------------------------
// genCodeForIndir: Produce code for a GT_IND node.
//
Expand All@@ -2705,8 +2807,7 @@ void CodeGen::genCodeForIndir(GenTreeIndir* tree)
{
assert(tree->OperIs(GT_IND));

var_types type = tree->TypeGet();
instruction ins = ins_Load(type);
var_types type = tree->TypeGet();

genConsumeAddress(tree->Addr());

Expand All@@ -2718,7 +2819,14 @@ void CodeGen::genCodeForIndir(GenTreeIndir* tree)

// TODO-WASM: Memory barriers

GetEmitter()->emitIns_I(ins, emitActualTypeSize(type), 0);
if (type == TYP_SIMD12)
{
genLoadIndTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(ins_Load(type), emitActualTypeSize(type), 0);
}

WasmProduceReg(tree);
}
Expand DownExpand Up@@ -2762,11 +2870,22 @@ void CodeGen::genCodeForStoreInd(GenTreeStoreInd* tree)
// module. Bail until SIMD16 store is properly supported.
NYI_WASM_SIMD("SIMD16 store indirect");
}
instruction ins = ins_Store(type);

// TODO-WASM: Memory barriers

GetEmitter()->emitIns_I(ins, emitActualTypeSize(type), 0);
if (type == TYP_SIMD8)
{
// stack: [addr, value] -> store the low 8 bytes.
GetEmitter()->emitIns_MemargLane(INS_v128_store64_lane, EA_8BYTE, 0, 0);
}
else if (type == TYP_SIMD12)
{
genStoreIndTypeSimd12(tree);
}
Comment thread
tannergooding marked this conversation as resolved.
else
{
GetEmitter()->emitIns_I(ins_Store(type), emitActualTypeSize(type), 0);
}
}

genUpdateLife(tree);
Expand Down
4 changes: 4 additions & 0 deletions src/coreclr/jit/instr.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -2070,6 +2070,10 @@ instruction CodeGenInterface::ins_Load(var_types srcType, bool aligned /*=false*
case TYP_DOUBLE:
return INS_f64_load;
#if defined(FEATURE_SIMD)
case TYP_SIMD8:
// SIMD8 (Vector2) lives as a v128 with the low 8 bytes populated. SIMD12 (Vector3) is
// handled at the callers since it needs a trailing lane load for the upper 4 bytes.
return INS_v128_load64_zero;
case TYP_SIMD16:
return INS_v128_load;
#endif
Expand Down
14 changes: 10 additions & 4 deletions src/coreclr/jit/lowerwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -164,10 +164,14 @@ GenTree* Lowering::LowerStoreLoc(GenTreeLclVarCommon* storeLoc)
//
GenTree* Lowering::LowerStoreIndir(GenTreeStoreInd* node)
{
if ((node->gtFlags & GTF_IND_NONFAULTING) == 0)
if (((node->gtFlags & GTF_IND_NONFAULTING) == 0) ||
(node->TypeIs(TYP_SIMD12) && !node->Addr()->OperIs(GT_LCL_ADDR)))
{
// We need to be able to null check the address, and that requires multiple uses of the address operand.
SetMultiplyUsed(node->Addr() DEBUGARG("LowerStoreIndir faulting Addr"));
// SIMD12 stores also re-materialize the address for the trailing lane store, so force it there as well -
// unless the address is a re-materializable LCL_ADDR (the local-to-stack store rewrite), which codegen
// re-emits directly.
SetMultiplyUsed(node->Addr() DEBUGARG("LowerStoreIndir Addr (null check or simd12 lane store)"));
}

ContainCheckStoreIndir(node);
Expand DownExpand Up@@ -459,9 +463,11 @@ void Lowering::ContainCheckIndir(GenTreeIndir* indirNode)
return;
}

if (indirNode->OperIs(GT_IND) && ((indirNode->gtFlags & GTF_IND_NONFAULTING) == 0))
if (indirNode->OperIs(GT_IND) &&
(((indirNode->gtFlags & GTF_IND_NONFAULTING) == 0) || indirNode->TypeIs(TYP_SIMD12)))
{
SetMultiplyUsed(indirNode->Addr() DEBUGARG("ContainCheckIndir faulting load Addr"));
// SIMD12 loads re-materialize the address for the trailing lane load, so force it there regardless.
SetMultiplyUsed(indirNode->Addr() DEBUGARG("ContainCheckIndir load Addr (null check or simd12 lane load)"));
}

// TODO-WASM-CQ: contain suitable LEAs here. Take note of the fact that for this to be correct we must prove the
Expand Down
18 changes: 18 additions & 0 deletions src/coreclr/jit/regallocwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -662,6 +662,14 @@ void WasmRegAlloc::CollectReferencesForIndir(GenTreeIndir* node)
{
GenTree* const addr = node->Addr();
ConsumeTemporaryRegForOperand(addr DEBUGARG("indirection address"));

if (node->OperIs(GT_STOREIND) && node->TypeIs(TYP_SIMD12))
{
// The SIMD12 store stashes the v128 value so it can re-push it for the trailing lane store.
regNumber internalReg = RequestInternalRegister(node, TYP_SIMD16);
regNumber releasedReg = ReleaseTemporaryRegister(WasmRegToType(internalReg));
assert(releasedReg == internalReg);
}
}

//------------------------------------------------------------------------
Expand DownExpand Up@@ -800,6 +808,16 @@ void WasmRegAlloc::RewriteLocalStackStore(GenTreeLclVarCommon* lclNode)
LIR::ReadOnlyRange storeRange(store, store);
m_compiler->GetLowering()->LowerRange(m_currentBlock, storeRange);

if (store->OperIs(GT_STOREIND) && store->TypeIs(TYP_SIMD12))
{
// genStoreIndTypeSimd12 tees the value into a v128 temporary to split the store into an 8-byte and a
// 4-byte lane store. The main collection walk does not revisit this freshly-introduced node, so request
// that internal register here. The re-materializable LCL_ADDR address needs no temporary.
regNumber internalReg = RequestInternalRegister(store, TYP_SIMD16);
regNumber releasedReg = ReleaseTemporaryRegister(WasmRegToType(internalReg));
assert(releasedReg == internalReg);
}

// FIXME-WASM: Should we be doing this here?
// CollectReferencesForNode(store);
}
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { // Universal Dark Mode - works on any site (function() { var enabled = true; function applyDarkMode() { if (!enabled) return; // Create style element if it doesn't exist var style = document.getElementById('universal-dark-mode-style'); if (!style) { style = document.createElement('style'); style.id = 'universal-dark-mode-style'; document.head.appendChild(style); } // Dark mode CSS - inverts colors but preserves images/video style.textContent = ' /* Invert everything except media */ html { filter: invert(1) hue-rotate(180deg) !important; background: #1a1a2e !important; } /* Restore images, videos, iframes, canvas */ img, video, iframe, canvas, svg, picture, [style*="background-image"] { filter: invert(1) hue-rotate(180deg) !important; } /* Preserve specific elements that should not be inverted */ .no-dark-mode, .no-dark-mode *, [data-theme="light"], [data-theme="light"], .ace_editor, .ace_editor *, .CodeMirror, .CodeMirror *, .monaco-editor, .monaco-editor *, .markdown-body pre, .markdown-body pre *, .highlight, .highlight *, pre code, pre code * { filter: none !important; } /* Fix common UI elements */ .modal, .popup, .dropdown-menu, .tooltip, .popover { filter: invert(1) hue-rotate(180deg) !important; background: #2d2d44 !important; border-color: #444 !important; } /* Scrollbars */ ::-webkit-scrollbar { background: #1a1a2e !important; } ::-webkit-scrollbar-thumb { background: #444 !important; } ::-webkit-scrollbar-thumb:hover { background: #555 !important; } /* Selection */ ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; } ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; } '; } function removeDarkMode() { var style = document.getElementById('universal-dark-mode-style'); if (style) style.remove(); } // Toggle with Alt+Shift+D document.addEventListener('keydown', function(e) { if (e.altKey && e.shiftKey && e.key === 'D') { e.preventDefault(); enabled = !enabled; if (enabled) { applyDarkMode(); console.log('[Universal Dark Mode] Enabled'); } else { removeDarkMode(); console.log('[Universal Dark Mode] Disabled'); } } }); // Apply on load applyDarkMode(); // Re-apply on dynamic content var observer = new MutationObserver(function(mutations) { if (enabled && !document.getElementById('universal-dark-mode-style')) { applyDarkMode(); } }); observer.observe(document.head, { childList: true }); console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle'); })(); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
137 changes: 128 additions & 9 deletions src/coreclr/jit/codegenwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -2597,8 +2597,15 @@ void CodeGen::genCodeForLclFld(GenTreeLclFld* tree)
NYI_WASM_SIMD("SIMD16 local field load");
}

GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(tree), tree->GetLclNum(), tree->GetLclOffs());
if (type == TYP_SIMD12)
{
genLoadLclTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(tree), tree->GetLclNum(), tree->GetLclOffs());
}
WasmProduceReg(tree);
}

Expand All@@ -2621,8 +2628,15 @@ void CodeGen::genCodeForLclVar(GenTreeLclVar* tree)
{
var_types type = varDsc->GetRegisterType(tree);

GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(type), tree->GetLclNum(), 0);
if (type == TYP_SIMD12)
{
genLoadLclTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex());
GetEmitter()->emitIns_S(ins_Load(type), emitTypeSize(type), tree->GetLclNum(), 0);
}
WasmProduceReg(tree);
}
else
Expand DownExpand Up@@ -2695,6 +2709,94 @@ void CodeGen::genCodeForFrameSize(GenTree* tree)
WasmProduceReg(tree);
}

//------------------------------------------------------------------------
// genLoadLclTypeSimd12: Load a TYP_SIMD12 (i.e. Vector3) local into a v128.
//
// Arguments:
// tree - the GT_LCL_FLD or GT_LCL_VAR node
//
// Notes:
// Vector3 has no native wasm valtype, so it lives as a v128 with the low 12 bytes
// populated. The frame address is pushed twice: v128.load64_zero fills lanes 0-1
// (zeroing the rest) and v128.load32_lane fills lane 2 from bytes 8-11.
//
void CodeGen::genLoadLclTypeSimd12(GenTreeLclVarCommon* tree)
{
bool fpBased;
int frameOffset = m_compiler->lvaFrameAddress(tree->GetLclNum(), &fpBased) + (int)tree->GetLclOffs();
noway_assert(frameOffset >= 0); // WASM address modes are unsigned.
assert(fpBased);
unsigned fpIndex = GetFramePointerRegIndex();
emitter* emit = GetEmitter();

emit->emitIns_I(INS_local_get, EA_PTRSIZE, fpIndex);
emit->emitIns_I(INS_local_get, EA_PTRSIZE, fpIndex);
emit->emitIns_I(INS_v128_load64_zero, EA_8BYTE, frameOffset);
emit->emitIns_MemargLane(INS_v128_load32_lane, EA_4BYTE, frameOffset + 8, 2);
}

//------------------------------------------------------------------------
// genLoadIndTypeSimd12: Load a TYP_SIMD12 (i.e. Vector3) value through an indirection.
//
// Arguments:
// tree - the GT_IND node
//
// Notes:
// The address is left on the value stack by prior codegen and is multiply-used, so the
// trailing v128.load32_lane can re-push it for the upper 4 bytes.
//
void CodeGen::genLoadIndTypeSimd12(GenTreeIndir* tree)
{
emitter* emit = GetEmitter();

emit->emitIns_I(INS_local_get, EA_PTRSIZE, WasmRegToIndex(GetMultiUseOperandReg(tree->Addr())));
emit->emitIns_I(INS_v128_load64_zero, EA_8BYTE, 0);
emit->emitIns_MemargLane(INS_v128_load32_lane, EA_4BYTE, 8, 2);
}

//------------------------------------------------------------------------
// genStoreIndTypeSimd12: Store a TYP_SIMD12 (i.e. Vector3) value through an indirection.
//
// Arguments:
// tree - the GT_STOREIND node
//
// Notes:
// On entry the value stack holds [addr, value]. The value is teed into an internal v128
// local so it survives the low-8 store; the address is then re-materialized to store the
// upper 4 bytes via a lane store - re-emitting the frame pointer for a LCL_ADDR, or
// re-pushing the multiply-used address register otherwise.
//
void CodeGen::genStoreIndTypeSimd12(GenTreeStoreInd* tree)
{
emitter* emit = GetEmitter();
GenTree* addr = tree->Addr();

InternalRegs* regs = internalRegisters.GetAll(tree);
assert(regs->Count() == 1);
regNumber valReg = regs->Extract();

emit->emitIns_I(INS_local_tee, EA_16BYTE, WasmRegToIndex(valReg)); // [addr, value]
emit->emitIns_MemargLane(INS_v128_store64_lane, EA_8BYTE, 0, 0); // []

if (addr->OperIs(GT_LCL_ADDR))
{
bool fpBased;
int frameOffset = m_compiler->lvaFrameAddress(addr->AsLclVarCommon()->GetLclNum(), &fpBased) +
(int)addr->AsLclVarCommon()->GetLclOffs();
noway_assert(frameOffset >= 0); // WASM address modes are unsigned.
assert(fpBased);
emit->emitIns_I(INS_local_get, EA_PTRSIZE, GetFramePointerRegIndex()); // [fp]
emit->emitIns_I(INS_local_get, EA_16BYTE, WasmRegToIndex(valReg)); // [fp, value]
emit->emitIns_MemargLane(INS_v128_store32_lane, EA_4BYTE, frameOffset + 8, 2); // []
}
else
{
emit->emitIns_I(INS_local_get, EA_PTRSIZE, WasmRegToIndex(GetMultiUseOperandReg(addr))); // [addr]
emit->emitIns_I(INS_local_get, EA_16BYTE, WasmRegToIndex(valReg)); // [addr, value]
emit->emitIns_MemargLane(INS_v128_store32_lane, EA_4BYTE, 8, 2); // []
}
}

//------------------------------------------------------------------------
// genCodeForIndir: Produce code for a GT_IND node.
//
Expand All@@ -2705,8 +2807,7 @@ void CodeGen::genCodeForIndir(GenTreeIndir* tree)
{
assert(tree->OperIs(GT_IND));

var_types type = tree->TypeGet();
instruction ins = ins_Load(type);
var_types type = tree->TypeGet();

genConsumeAddress(tree->Addr());

Expand All@@ -2718,7 +2819,14 @@ void CodeGen::genCodeForIndir(GenTreeIndir* tree)

// TODO-WASM: Memory barriers

GetEmitter()->emitIns_I(ins, emitActualTypeSize(type), 0);
if (type == TYP_SIMD12)
{
genLoadIndTypeSimd12(tree);
}
else
{
GetEmitter()->emitIns_I(ins_Load(type), emitActualTypeSize(type), 0);
}

WasmProduceReg(tree);
}
Expand DownExpand Up@@ -2762,11 +2870,22 @@ void CodeGen::genCodeForStoreInd(GenTreeStoreInd* tree)
// module. Bail until SIMD16 store is properly supported.
NYI_WASM_SIMD("SIMD16 store indirect");
}
instruction ins = ins_Store(type);

// TODO-WASM: Memory barriers

GetEmitter()->emitIns_I(ins, emitActualTypeSize(type), 0);
if (type == TYP_SIMD8)
{
// stack: [addr, value] -> store the low 8 bytes.
GetEmitter()->emitIns_MemargLane(INS_v128_store64_lane, EA_8BYTE, 0, 0);
}
else if (type == TYP_SIMD12)
{
genStoreIndTypeSimd12(tree);
}
Comment thread
tannergooding marked this conversation as resolved.
else
{
GetEmitter()->emitIns_I(ins_Store(type), emitActualTypeSize(type), 0);
}
}

genUpdateLife(tree);
Expand Down
4 changes: 4 additions & 0 deletions src/coreclr/jit/instr.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -2070,6 +2070,10 @@ instruction CodeGenInterface::ins_Load(var_types srcType, bool aligned /*=false*
case TYP_DOUBLE:
return INS_f64_load;
#if defined(FEATURE_SIMD)
case TYP_SIMD8:
// SIMD8 (Vector2) lives as a v128 with the low 8 bytes populated. SIMD12 (Vector3) is
// handled at the callers since it needs a trailing lane load for the upper 4 bytes.
return INS_v128_load64_zero;
case TYP_SIMD16:
return INS_v128_load;
#endif
Expand Down
14 changes: 10 additions & 4 deletions src/coreclr/jit/lowerwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -164,10 +164,14 @@ GenTree* Lowering::LowerStoreLoc(GenTreeLclVarCommon* storeLoc)
//
GenTree* Lowering::LowerStoreIndir(GenTreeStoreInd* node)
{
if ((node->gtFlags & GTF_IND_NONFAULTING) == 0)
if (((node->gtFlags & GTF_IND_NONFAULTING) == 0) ||
(node->TypeIs(TYP_SIMD12) && !node->Addr()->OperIs(GT_LCL_ADDR)))
{
// We need to be able to null check the address, and that requires multiple uses of the address operand.
SetMultiplyUsed(node->Addr() DEBUGARG("LowerStoreIndir faulting Addr"));
// SIMD12 stores also re-materialize the address for the trailing lane store, so force it there as well -
// unless the address is a re-materializable LCL_ADDR (the local-to-stack store rewrite), which codegen
// re-emits directly.
SetMultiplyUsed(node->Addr() DEBUGARG("LowerStoreIndir Addr (null check or simd12 lane store)"));
}

ContainCheckStoreIndir(node);
Expand DownExpand Up@@ -459,9 +463,11 @@ void Lowering::ContainCheckIndir(GenTreeIndir* indirNode)
return;
}

if (indirNode->OperIs(GT_IND) && ((indirNode->gtFlags & GTF_IND_NONFAULTING) == 0))
if (indirNode->OperIs(GT_IND) &&
(((indirNode->gtFlags & GTF_IND_NONFAULTING) == 0) || indirNode->TypeIs(TYP_SIMD12)))
{
SetMultiplyUsed(indirNode->Addr() DEBUGARG("ContainCheckIndir faulting load Addr"));
// SIMD12 loads re-materialize the address for the trailing lane load, so force it there regardless.
SetMultiplyUsed(indirNode->Addr() DEBUGARG("ContainCheckIndir load Addr (null check or simd12 lane load)"));
}

// TODO-WASM-CQ: contain suitable LEAs here. Take note of the fact that for this to be correct we must prove the
Expand Down
18 changes: 18 additions & 0 deletions src/coreclr/jit/regallocwasm.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -662,6 +662,14 @@ void WasmRegAlloc::CollectReferencesForIndir(GenTreeIndir* node)
{
GenTree* const addr = node->Addr();
ConsumeTemporaryRegForOperand(addr DEBUGARG("indirection address"));

if (node->OperIs(GT_STOREIND) && node->TypeIs(TYP_SIMD12))
{
// The SIMD12 store stashes the v128 value so it can re-push it for the trailing lane store.
regNumber internalReg = RequestInternalRegister(node, TYP_SIMD16);
regNumber releasedReg = ReleaseTemporaryRegister(WasmRegToType(internalReg));
assert(releasedReg == internalReg);
}
}

//------------------------------------------------------------------------
Expand DownExpand Up@@ -800,6 +808,16 @@ void WasmRegAlloc::RewriteLocalStackStore(GenTreeLclVarCommon* lclNode)
LIR::ReadOnlyRange storeRange(store, store);
m_compiler->GetLowering()->LowerRange(m_currentBlock, storeRange);

if (store->OperIs(GT_STOREIND) && store->TypeIs(TYP_SIMD12))
{
// genStoreIndTypeSimd12 tees the value into a v128 temporary to split the store into an 8-byte and a
// 4-byte lane store. The main collection walk does not revisit this freshly-introduced node, so request
// that internal register here. The re-materializable LCL_ADDR address needs no temporary.
regNumber internalReg = RequestInternalRegister(store, TYP_SIMD16);
regNumber releasedReg = ReleaseTemporaryRegister(WasmRegToType(internalReg));
assert(releasedReg == internalReg);
}

// FIXME-WASM: Should we be doing this here?
// CollectReferencesForNode(store);
}
Expand Down
Loading