Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
26 changes: 26 additions & 0 deletions src/coreclr/jit/simd.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -440,6 +440,12 @@ TBase EvaluateBinaryScalarRSZ(TBase arg0, TBase arg1)
return arg0 >> (arg1 & ((sizeof(TBase) * 8) - 1));
}

template <typename TBase>
TBase GetAllBitsSetScalar()
{
return ~static_cast<TBase>(0);
}

template <>
inline int8_t EvaluateBinaryScalarRSZ<int8_t>(int8_t arg0, int8_t arg1)
{
Expand DownExpand Up@@ -520,6 +526,26 @@ TBase EvaluateBinaryScalarSpecialized(genTreeOps oper, TBase arg0, TBase arg1)
return arg0 ^ arg1;
}

case GT_EQ:
{
#ifdef _MSC_VER
// Floating point is not supported
assert(&typeid(TBase) != &typeid(float));
assert(&typeid(TBase) != &typeid(double));
#endif // _MSC_VER
return arg0 == arg1 ? GetAllBitsSetScalar<TBase>() : 0;
}

case GT_NE:
{
#ifdef _MSC_VER
// Floating point is not supported
assert(&typeid(TBase) != &typeid(float));
assert(&typeid(TBase) != &typeid(double));
#endif // _MSC_VER
return arg0 != arg1 ? GetAllBitsSetScalar<TBase>() : 0;
}

default:
{
unreached();
Expand Down
111 changes: 105 additions & 6 deletions src/coreclr/jit/valuenum.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -6812,6 +6812,47 @@ ValueNum EvaluateBinarySimd(ValueNumStore* vns,
}
#endif // TARGET_XARCH

case TYP_BOOL:
{
assert((oper == GT_EQ) || (oper == GT_NE));

var_types vn1Type = vns->TypeOfVN(arg0VN);
var_types vn2Type = vns->TypeOfVN(arg1VN);

assert((vn1Type == vn2Type) && varTypeIsSIMD(vn1Type));
assert(!varTypeIsFloating(baseType));

ValueNum packed = EvaluateBinarySimd(vns, GT_EQ, scalar, vn1Type, baseType, arg0VN, arg1VN);

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

shouldn't this be EvaluateBinarySimd(vns, oper, ...)?

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I wonder if it would be simpler to implement this as EvaluateVector and then simplify check result IsAllBitsSet or !IsZero

Which would also simplify the other relational comparisons

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

shouldn't this be EvaluateBinarySimd(vns, oper, ...)?

It makes it harder to check output, doesn't it?
Currently I just pass EQ and then return AllBitsSet or !AllBitsSet depending on source oper

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Does it? We have a simple property for any simd_T and it makes it easier to cover the other cases.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

That is, if you just evaluate per element you get a result simd_T result and can then just check IsAllBitsSet or IsZero

bool allBitsSet = false;
if (vn1Type == TYP_SIMD8)
{
allBitsSet = GetConstantSimd8(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD12)
{
allBitsSet = GetConstantSimd12(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD16)
{
allBitsSet = GetConstantSimd16(vns, baseType, packed).IsAllBitsSet();
}
#if defined(TARGET_XARCH)
else if (vn1Type == TYP_SIMD32)
{
allBitsSet = GetConstantSimd32(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD64)
{
allBitsSet = GetConstantSimd64(vns, baseType, packed).IsAllBitsSet();
}
#endif
else
{
unreached();
}
return vns->VNForIntCon(oper == GT_EQ ? allBitsSet : !allBitsSet);
}

default:
{
unreached();
Expand DownExpand Up@@ -7167,6 +7208,43 @@ ValueNum ValueNumStore::EvalHWIntrinsicFunBinary(var_types type,

switch (ni)
{
#ifdef TARGET_ARM64
case NI_Vector64_op_Equality:
case NI_Vector128_op_Equality:
case NI_Vector64_EqualsAll:
case NI_Vector128_EqualsAll:
#else
case NI_Vector128_op_Equality:
case NI_Vector256_op_Equality:
case NI_Vector512_op_Equality:
case NI_Vector128_EqualsAll:
case NI_Vector256_EqualsAll:
case NI_Vector512_EqualsAll:
#endif
{
if (!varTypeIsFloating(baseType))
{
return EvaluateBinarySimd(this, GT_EQ, /* scalar */ false, type, baseType, arg0VN, arg1VN);
}
break;
}

#ifdef TARGET_ARM64
case NI_Vector64_op_Inequality:
case NI_Vector128_op_Inequality:
#else
case NI_Vector128_op_Inequality:
case NI_Vector256_op_Inequality:
case NI_Vector512_op_Inequality:
#endif
{
if (!varTypeIsFloating(baseType))
{
return EvaluateBinarySimd(this, GT_NE, /* scalar */ false, type, baseType, arg0VN, arg1VN);
}
break;
}

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Do we want to handle GT, LT, GE, and LE simultaneously?

Likewise given you've added the support for computing the vector version in order to make bool work, should we just handle the intrinsics that produce a vector as well?

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Those had no hits so I just didn't want to add more code and tests 🙂 Although, even EQ/NE don't have hits, I just found a use case outside.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think this is the type of scenario where it’s a core comparison on SIMD and so we want to generally handle it even if our own code doesn’t have any hits today.

it’s odd to cover some of the relational comparisons and not the others


#ifdef TARGET_ARM64
case NI_AdvSimd_Add:
case NI_AdvSimd_Arm64_Add:
Expand DownExpand Up@@ -10269,14 +10347,35 @@ void Compiler::fgValueNumberSsaVarDef(GenTreeLclVarCommon* lcl)
static bool GetStaticFieldSeqAndAddress(ValueNumStore* vnStore, GenTree* tree, ssize_t* byteOffset, FieldSeq** pFseq)
{
VNFuncApp funcApp;
if (vnStore->GetVNFunc(tree->gtVNPair.GetLiberal(), &funcApp) && (funcApp.m_func == VNF_PtrToStatic))
if (vnStore->GetVNFunc(tree->gtVNPair.GetLiberal(), &funcApp))
{
FieldSeq* fseq = vnStore->FieldSeqVNToFieldSeq(funcApp.m_args[1]);
if (fseq->GetKind() == FieldSeq::FieldKind::SimpleStatic)
if (funcApp.m_func == VNF_PtrToStatic)
{
*byteOffset = vnStore->ConstantValue<ssize_t>(funcApp.m_args[2]);
*pFseq = fseq;
return true;
FieldSeq* fseq = vnStore->FieldSeqVNToFieldSeq(funcApp.m_args[1]);
if (fseq->GetKind() == FieldSeq::FieldKind::SimpleStatic)
{
*byteOffset = vnStore->ConstantValue<ssize_t>(funcApp.m_args[2]);
*pFseq = fseq;
return true;
}
}
else if (funcApp.m_func == VNFunc(GT_ADD))
{
// Handle ADD(STATIC_HDL, OFFSET) via VN (the logic in this method mostly works with plain tree nodes)
if (vnStore->IsVNHandle(funcApp.m_args[0]) &&
(vnStore->GetHandleFlags(funcApp.m_args[0]) == GTF_ICON_STATIC_HDL) &&
vnStore->IsVNConstant(funcApp.m_args[1]) && !vnStore->IsVNHandle(funcApp.m_args[1]))
{
FieldSeq* fldSeq = vnStore->GetFieldSeqFromAddress(funcApp.m_args[0]);
if (fldSeq != nullptr)
{
assert(fldSeq->GetKind() == FieldSeq::FieldKind::SimpleStaticKnownAddress);
*pFseq = fldSeq;
*byteOffset = vnStore->CoercedConstantValue<ssize_t>(funcApp.m_args[0]) - fldSeq->GetOffset() +
vnStore->CoercedConstantValue<ssize_t>(funcApp.m_args[1]);
return true;
}
}
}
}
ssize_t val = 0;
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -733,4 +733,34 @@ public static void XorTests()
^ Vector128.Create((double)(+1), +1)
);
}

[Fact]
public static void EqualityTests()
{
Assert.True(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0) ==
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.False(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0) !=
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.False(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xFF) ==
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.True(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xFF) !=
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.True(
Vector128.EqualsAll(
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8),
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8)));
Assert.False(
Vector128.EqualsAll(
Vector128.Create((ushort)(0x0001), 0xFFFF, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8),
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8)));
Assert.Equal(
Vector128.Create(4294967295, 0, 4294967295, 4294967295),
Vector128.Equals(
Vector128.Create((uint)(0x0000_0001), 0xFFFF_FFFE, 0x0000_0003, 0xFFFF_FFFC),
Vector128.Create((uint)(0x0000_0001), 0x0000_0001, 0x0000_0003, 0xFFFF_FFFC)));
}
}
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
26 changes: 26 additions & 0 deletions src/coreclr/jit/simd.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -440,6 +440,12 @@ TBase EvaluateBinaryScalarRSZ(TBase arg0, TBase arg1)
return arg0 >> (arg1 & ((sizeof(TBase) * 8) - 1));
}

template <typename TBase>
TBase GetAllBitsSetScalar()
{
return ~static_cast<TBase>(0);
}

template <>
inline int8_t EvaluateBinaryScalarRSZ<int8_t>(int8_t arg0, int8_t arg1)
{
Expand DownExpand Up@@ -520,6 +526,26 @@ TBase EvaluateBinaryScalarSpecialized(genTreeOps oper, TBase arg0, TBase arg1)
return arg0 ^ arg1;
}

case GT_EQ:
{
#ifdef _MSC_VER
// Floating point is not supported
assert(&typeid(TBase) != &typeid(float));
assert(&typeid(TBase) != &typeid(double));
#endif // _MSC_VER
return arg0 == arg1 ? GetAllBitsSetScalar<TBase>() : 0;
}

case GT_NE:
{
#ifdef _MSC_VER
// Floating point is not supported
assert(&typeid(TBase) != &typeid(float));
assert(&typeid(TBase) != &typeid(double));
#endif // _MSC_VER
return arg0 != arg1 ? GetAllBitsSetScalar<TBase>() : 0;
}

default:
{
unreached();
Expand Down
111 changes: 105 additions & 6 deletions src/coreclr/jit/valuenum.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -6812,6 +6812,47 @@ ValueNum EvaluateBinarySimd(ValueNumStore* vns,
}
#endif // TARGET_XARCH

case TYP_BOOL:
{
assert((oper == GT_EQ) || (oper == GT_NE));

var_types vn1Type = vns->TypeOfVN(arg0VN);
var_types vn2Type = vns->TypeOfVN(arg1VN);

assert((vn1Type == vn2Type) && varTypeIsSIMD(vn1Type));
assert(!varTypeIsFloating(baseType));

ValueNum packed = EvaluateBinarySimd(vns, GT_EQ, scalar, vn1Type, baseType, arg0VN, arg1VN);

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

shouldn't this be EvaluateBinarySimd(vns, oper, ...)?

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I wonder if it would be simpler to implement this as EvaluateVector and then simplify check result IsAllBitsSet or !IsZero

Which would also simplify the other relational comparisons

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

shouldn't this be EvaluateBinarySimd(vns, oper, ...)?

It makes it harder to check output, doesn't it?
Currently I just pass EQ and then return AllBitsSet or !AllBitsSet depending on source oper

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Does it? We have a simple property for any simd_T and it makes it easier to cover the other cases.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

That is, if you just evaluate per element you get a result simd_T result and can then just check IsAllBitsSet or IsZero

bool allBitsSet = false;
if (vn1Type == TYP_SIMD8)
{
allBitsSet = GetConstantSimd8(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD12)
{
allBitsSet = GetConstantSimd12(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD16)
{
allBitsSet = GetConstantSimd16(vns, baseType, packed).IsAllBitsSet();
}
#if defined(TARGET_XARCH)
else if (vn1Type == TYP_SIMD32)
{
allBitsSet = GetConstantSimd32(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD64)
{
allBitsSet = GetConstantSimd64(vns, baseType, packed).IsAllBitsSet();
}
#endif
else
{
unreached();
}
return vns->VNForIntCon(oper == GT_EQ ? allBitsSet : !allBitsSet);
}

default:
{
unreached();
Expand DownExpand Up@@ -7167,6 +7208,43 @@ ValueNum ValueNumStore::EvalHWIntrinsicFunBinary(var_types type,

switch (ni)
{
#ifdef TARGET_ARM64
case NI_Vector64_op_Equality:
case NI_Vector128_op_Equality:
case NI_Vector64_EqualsAll:
case NI_Vector128_EqualsAll:
#else
case NI_Vector128_op_Equality:
case NI_Vector256_op_Equality:
case NI_Vector512_op_Equality:
case NI_Vector128_EqualsAll:
case NI_Vector256_EqualsAll:
case NI_Vector512_EqualsAll:
#endif
{
if (!varTypeIsFloating(baseType))
{
return EvaluateBinarySimd(this, GT_EQ, /* scalar */ false, type, baseType, arg0VN, arg1VN);
}
break;
}

#ifdef TARGET_ARM64
case NI_Vector64_op_Inequality:
case NI_Vector128_op_Inequality:
#else
case NI_Vector128_op_Inequality:
case NI_Vector256_op_Inequality:
case NI_Vector512_op_Inequality:
#endif
{
if (!varTypeIsFloating(baseType))
{
return EvaluateBinarySimd(this, GT_NE, /* scalar */ false, type, baseType, arg0VN, arg1VN);
}
break;
}

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Do we want to handle GT, LT, GE, and LE simultaneously?

Likewise given you've added the support for computing the vector version in order to make bool work, should we just handle the intrinsics that produce a vector as well?

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Those had no hits so I just didn't want to add more code and tests 🙂 Although, even EQ/NE don't have hits, I just found a use case outside.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think this is the type of scenario where it’s a core comparison on SIMD and so we want to generally handle it even if our own code doesn’t have any hits today.

it’s odd to cover some of the relational comparisons and not the others


#ifdef TARGET_ARM64
case NI_AdvSimd_Add:
case NI_AdvSimd_Arm64_Add:
Expand DownExpand Up@@ -10269,14 +10347,35 @@ void Compiler::fgValueNumberSsaVarDef(GenTreeLclVarCommon* lcl)
static bool GetStaticFieldSeqAndAddress(ValueNumStore* vnStore, GenTree* tree, ssize_t* byteOffset, FieldSeq** pFseq)
{
VNFuncApp funcApp;
if (vnStore->GetVNFunc(tree->gtVNPair.GetLiberal(), &funcApp) && (funcApp.m_func == VNF_PtrToStatic))
if (vnStore->GetVNFunc(tree->gtVNPair.GetLiberal(), &funcApp))
{
FieldSeq* fseq = vnStore->FieldSeqVNToFieldSeq(funcApp.m_args[1]);
if (fseq->GetKind() == FieldSeq::FieldKind::SimpleStatic)
if (funcApp.m_func == VNF_PtrToStatic)
{
*byteOffset = vnStore->ConstantValue<ssize_t>(funcApp.m_args[2]);
*pFseq = fseq;
return true;
FieldSeq* fseq = vnStore->FieldSeqVNToFieldSeq(funcApp.m_args[1]);
if (fseq->GetKind() == FieldSeq::FieldKind::SimpleStatic)
{
*byteOffset = vnStore->ConstantValue<ssize_t>(funcApp.m_args[2]);
*pFseq = fseq;
return true;
}
}
else if (funcApp.m_func == VNFunc(GT_ADD))
{
// Handle ADD(STATIC_HDL, OFFSET) via VN (the logic in this method mostly works with plain tree nodes)
if (vnStore->IsVNHandle(funcApp.m_args[0]) &&
(vnStore->GetHandleFlags(funcApp.m_args[0]) == GTF_ICON_STATIC_HDL) &&
vnStore->IsVNConstant(funcApp.m_args[1]) && !vnStore->IsVNHandle(funcApp.m_args[1]))
{
FieldSeq* fldSeq = vnStore->GetFieldSeqFromAddress(funcApp.m_args[0]);
if (fldSeq != nullptr)
{
assert(fldSeq->GetKind() == FieldSeq::FieldKind::SimpleStaticKnownAddress);
*pFseq = fldSeq;
*byteOffset = vnStore->CoercedConstantValue<ssize_t>(funcApp.m_args[0]) - fldSeq->GetOffset() +
vnStore->CoercedConstantValue<ssize_t>(funcApp.m_args[1]);
return true;
}
}
}
}
ssize_t val = 0;
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -733,4 +733,34 @@ public static void XorTests()
^ Vector128.Create((double)(+1), +1)
);
}

[Fact]
public static void EqualityTests()
{
Assert.True(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0) ==
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.False(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0) !=
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.False(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xFF) ==
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.True(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xFF) !=
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.True(
Vector128.EqualsAll(
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8),
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8)));
Assert.False(
Vector128.EqualsAll(
Vector128.Create((ushort)(0x0001), 0xFFFF, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8),
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8)));
Assert.Equal(
Vector128.Create(4294967295, 0, 4294967295, 4294967295),
Vector128.Equals(
Vector128.Create((uint)(0x0000_0001), 0xFFFF_FFFE, 0x0000_0003, 0xFFFF_FFFC),
Vector128.Create((uint)(0x0000_0001), 0x0000_0001, 0x0000_0003, 0xFFFF_FFFC)));
}
}
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
26 changes: 26 additions & 0 deletions src/coreclr/jit/simd.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -440,6 +440,12 @@ TBase EvaluateBinaryScalarRSZ(TBase arg0, TBase arg1)
return arg0 >> (arg1 & ((sizeof(TBase) * 8) - 1));
}

template <typename TBase>
TBase GetAllBitsSetScalar()
{
return ~static_cast<TBase>(0);
}

template <>
inline int8_t EvaluateBinaryScalarRSZ<int8_t>(int8_t arg0, int8_t arg1)
{
Expand DownExpand Up@@ -520,6 +526,26 @@ TBase EvaluateBinaryScalarSpecialized(genTreeOps oper, TBase arg0, TBase arg1)
return arg0 ^ arg1;
}

case GT_EQ:
{
#ifdef _MSC_VER
// Floating point is not supported
assert(&typeid(TBase) != &typeid(float));
assert(&typeid(TBase) != &typeid(double));
#endif // _MSC_VER
return arg0 == arg1 ? GetAllBitsSetScalar<TBase>() : 0;
}

case GT_NE:
{
#ifdef _MSC_VER
// Floating point is not supported
assert(&typeid(TBase) != &typeid(float));
assert(&typeid(TBase) != &typeid(double));
#endif // _MSC_VER
return arg0 != arg1 ? GetAllBitsSetScalar<TBase>() : 0;
}

default:
{
unreached();
Expand Down
111 changes: 105 additions & 6 deletions src/coreclr/jit/valuenum.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -6812,6 +6812,47 @@ ValueNum EvaluateBinarySimd(ValueNumStore* vns,
}
#endif // TARGET_XARCH

case TYP_BOOL:
{
assert((oper == GT_EQ) || (oper == GT_NE));

var_types vn1Type = vns->TypeOfVN(arg0VN);
var_types vn2Type = vns->TypeOfVN(arg1VN);

assert((vn1Type == vn2Type) && varTypeIsSIMD(vn1Type));
assert(!varTypeIsFloating(baseType));

ValueNum packed = EvaluateBinarySimd(vns, GT_EQ, scalar, vn1Type, baseType, arg0VN, arg1VN);

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

shouldn't this be EvaluateBinarySimd(vns, oper, ...)?

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I wonder if it would be simpler to implement this as EvaluateVector and then simplify check result IsAllBitsSet or !IsZero

Which would also simplify the other relational comparisons

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

shouldn't this be EvaluateBinarySimd(vns, oper, ...)?

It makes it harder to check output, doesn't it?
Currently I just pass EQ and then return AllBitsSet or !AllBitsSet depending on source oper

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Does it? We have a simple property for any simd_T and it makes it easier to cover the other cases.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

That is, if you just evaluate per element you get a result simd_T result and can then just check IsAllBitsSet or IsZero

bool allBitsSet = false;
if (vn1Type == TYP_SIMD8)
{
allBitsSet = GetConstantSimd8(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD12)
{
allBitsSet = GetConstantSimd12(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD16)
{
allBitsSet = GetConstantSimd16(vns, baseType, packed).IsAllBitsSet();
}
#if defined(TARGET_XARCH)
else if (vn1Type == TYP_SIMD32)
{
allBitsSet = GetConstantSimd32(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD64)
{
allBitsSet = GetConstantSimd64(vns, baseType, packed).IsAllBitsSet();
}
#endif
else
{
unreached();
}
return vns->VNForIntCon(oper == GT_EQ ? allBitsSet : !allBitsSet);
}

default:
{
unreached();
Expand DownExpand Up@@ -7167,6 +7208,43 @@ ValueNum ValueNumStore::EvalHWIntrinsicFunBinary(var_types type,

switch (ni)
{
#ifdef TARGET_ARM64
case NI_Vector64_op_Equality:
case NI_Vector128_op_Equality:
case NI_Vector64_EqualsAll:
case NI_Vector128_EqualsAll:
#else
case NI_Vector128_op_Equality:
case NI_Vector256_op_Equality:
case NI_Vector512_op_Equality:
case NI_Vector128_EqualsAll:
case NI_Vector256_EqualsAll:
case NI_Vector512_EqualsAll:
#endif
{
if (!varTypeIsFloating(baseType))
{
return EvaluateBinarySimd(this, GT_EQ, /* scalar */ false, type, baseType, arg0VN, arg1VN);
}
break;
}

#ifdef TARGET_ARM64
case NI_Vector64_op_Inequality:
case NI_Vector128_op_Inequality:
#else
case NI_Vector128_op_Inequality:
case NI_Vector256_op_Inequality:
case NI_Vector512_op_Inequality:
#endif
{
if (!varTypeIsFloating(baseType))
{
return EvaluateBinarySimd(this, GT_NE, /* scalar */ false, type, baseType, arg0VN, arg1VN);
}
break;
}

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Do we want to handle GT, LT, GE, and LE simultaneously?

Likewise given you've added the support for computing the vector version in order to make bool work, should we just handle the intrinsics that produce a vector as well?

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Those had no hits so I just didn't want to add more code and tests 🙂 Although, even EQ/NE don't have hits, I just found a use case outside.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think this is the type of scenario where it’s a core comparison on SIMD and so we want to generally handle it even if our own code doesn’t have any hits today.

it’s odd to cover some of the relational comparisons and not the others


#ifdef TARGET_ARM64
case NI_AdvSimd_Add:
case NI_AdvSimd_Arm64_Add:
Expand DownExpand Up@@ -10269,14 +10347,35 @@ void Compiler::fgValueNumberSsaVarDef(GenTreeLclVarCommon* lcl)
static bool GetStaticFieldSeqAndAddress(ValueNumStore* vnStore, GenTree* tree, ssize_t* byteOffset, FieldSeq** pFseq)
{
VNFuncApp funcApp;
if (vnStore->GetVNFunc(tree->gtVNPair.GetLiberal(), &funcApp) && (funcApp.m_func == VNF_PtrToStatic))
if (vnStore->GetVNFunc(tree->gtVNPair.GetLiberal(), &funcApp))
{
FieldSeq* fseq = vnStore->FieldSeqVNToFieldSeq(funcApp.m_args[1]);
if (fseq->GetKind() == FieldSeq::FieldKind::SimpleStatic)
if (funcApp.m_func == VNF_PtrToStatic)
{
*byteOffset = vnStore->ConstantValue<ssize_t>(funcApp.m_args[2]);
*pFseq = fseq;
return true;
FieldSeq* fseq = vnStore->FieldSeqVNToFieldSeq(funcApp.m_args[1]);
if (fseq->GetKind() == FieldSeq::FieldKind::SimpleStatic)
{
*byteOffset = vnStore->ConstantValue<ssize_t>(funcApp.m_args[2]);
*pFseq = fseq;
return true;
}
}
else if (funcApp.m_func == VNFunc(GT_ADD))
{
// Handle ADD(STATIC_HDL, OFFSET) via VN (the logic in this method mostly works with plain tree nodes)
if (vnStore->IsVNHandle(funcApp.m_args[0]) &&
(vnStore->GetHandleFlags(funcApp.m_args[0]) == GTF_ICON_STATIC_HDL) &&
vnStore->IsVNConstant(funcApp.m_args[1]) && !vnStore->IsVNHandle(funcApp.m_args[1]))
{
FieldSeq* fldSeq = vnStore->GetFieldSeqFromAddress(funcApp.m_args[0]);
if (fldSeq != nullptr)
{
assert(fldSeq->GetKind() == FieldSeq::FieldKind::SimpleStaticKnownAddress);
*pFseq = fldSeq;
*byteOffset = vnStore->CoercedConstantValue<ssize_t>(funcApp.m_args[0]) - fldSeq->GetOffset() +
vnStore->CoercedConstantValue<ssize_t>(funcApp.m_args[1]);
return true;
}
}
}
}
ssize_t val = 0;
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -733,4 +733,34 @@ public static void XorTests()
^ Vector128.Create((double)(+1), +1)
);
}

[Fact]
public static void EqualityTests()
{
Assert.True(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0) ==
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.False(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0) !=
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.False(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xFF) ==
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.True(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xFF) !=
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.True(
Vector128.EqualsAll(
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8),
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8)));
Assert.False(
Vector128.EqualsAll(
Vector128.Create((ushort)(0x0001), 0xFFFF, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8),
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8)));
Assert.Equal(
Vector128.Create(4294967295, 0, 4294967295, 4294967295),
Vector128.Equals(
Vector128.Create((uint)(0x0000_0001), 0xFFFF_FFFE, 0x0000_0003, 0xFFFF_FFFC),
Vector128.Create((uint)(0x0000_0001), 0x0000_0001, 0x0000_0003, 0xFFFF_FFFC)));
}
}
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
26 changes: 26 additions & 0 deletions src/coreclr/jit/simd.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -440,6 +440,12 @@ TBase EvaluateBinaryScalarRSZ(TBase arg0, TBase arg1)
return arg0 >> (arg1 & ((sizeof(TBase) * 8) - 1));
}

template <typename TBase>
TBase GetAllBitsSetScalar()
{
return ~static_cast<TBase>(0);
}

template <>
inline int8_t EvaluateBinaryScalarRSZ<int8_t>(int8_t arg0, int8_t arg1)
{
Expand DownExpand Up@@ -520,6 +526,26 @@ TBase EvaluateBinaryScalarSpecialized(genTreeOps oper, TBase arg0, TBase arg1)
return arg0 ^ arg1;
}

case GT_EQ:
{
#ifdef _MSC_VER
// Floating point is not supported
assert(&typeid(TBase) != &typeid(float));
assert(&typeid(TBase) != &typeid(double));
#endif // _MSC_VER
return arg0 == arg1 ? GetAllBitsSetScalar<TBase>() : 0;
}

case GT_NE:
{
#ifdef _MSC_VER
// Floating point is not supported
assert(&typeid(TBase) != &typeid(float));
assert(&typeid(TBase) != &typeid(double));
#endif // _MSC_VER
return arg0 != arg1 ? GetAllBitsSetScalar<TBase>() : 0;
}

default:
{
unreached();
Expand Down
111 changes: 105 additions & 6 deletions src/coreclr/jit/valuenum.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -6812,6 +6812,47 @@ ValueNum EvaluateBinarySimd(ValueNumStore* vns,
}
#endif // TARGET_XARCH

case TYP_BOOL:
{
assert((oper == GT_EQ) || (oper == GT_NE));

var_types vn1Type = vns->TypeOfVN(arg0VN);
var_types vn2Type = vns->TypeOfVN(arg1VN);

assert((vn1Type == vn2Type) && varTypeIsSIMD(vn1Type));
assert(!varTypeIsFloating(baseType));

ValueNum packed = EvaluateBinarySimd(vns, GT_EQ, scalar, vn1Type, baseType, arg0VN, arg1VN);

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

shouldn't this be EvaluateBinarySimd(vns, oper, ...)?

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I wonder if it would be simpler to implement this as EvaluateVector and then simplify check result IsAllBitsSet or !IsZero

Which would also simplify the other relational comparisons

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

shouldn't this be EvaluateBinarySimd(vns, oper, ...)?

It makes it harder to check output, doesn't it?
Currently I just pass EQ and then return AllBitsSet or !AllBitsSet depending on source oper

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Does it? We have a simple property for any simd_T and it makes it easier to cover the other cases.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

That is, if you just evaluate per element you get a result simd_T result and can then just check IsAllBitsSet or IsZero

bool allBitsSet = false;
if (vn1Type == TYP_SIMD8)
{
allBitsSet = GetConstantSimd8(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD12)
{
allBitsSet = GetConstantSimd12(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD16)
{
allBitsSet = GetConstantSimd16(vns, baseType, packed).IsAllBitsSet();
}
#if defined(TARGET_XARCH)
else if (vn1Type == TYP_SIMD32)
{
allBitsSet = GetConstantSimd32(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD64)
{
allBitsSet = GetConstantSimd64(vns, baseType, packed).IsAllBitsSet();
}
#endif
else
{
unreached();
}
return vns->VNForIntCon(oper == GT_EQ ? allBitsSet : !allBitsSet);
}

default:
{
unreached();
Expand DownExpand Up@@ -7167,6 +7208,43 @@ ValueNum ValueNumStore::EvalHWIntrinsicFunBinary(var_types type,

switch (ni)
{
#ifdef TARGET_ARM64
case NI_Vector64_op_Equality:
case NI_Vector128_op_Equality:
case NI_Vector64_EqualsAll:
case NI_Vector128_EqualsAll:
#else
case NI_Vector128_op_Equality:
case NI_Vector256_op_Equality:
case NI_Vector512_op_Equality:
case NI_Vector128_EqualsAll:
case NI_Vector256_EqualsAll:
case NI_Vector512_EqualsAll:
#endif
{
if (!varTypeIsFloating(baseType))
{
return EvaluateBinarySimd(this, GT_EQ, /* scalar */ false, type, baseType, arg0VN, arg1VN);
}
break;
}

#ifdef TARGET_ARM64
case NI_Vector64_op_Inequality:
case NI_Vector128_op_Inequality:
#else
case NI_Vector128_op_Inequality:
case NI_Vector256_op_Inequality:
case NI_Vector512_op_Inequality:
#endif
{
if (!varTypeIsFloating(baseType))
{
return EvaluateBinarySimd(this, GT_NE, /* scalar */ false, type, baseType, arg0VN, arg1VN);
}
break;
}

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Do we want to handle GT, LT, GE, and LE simultaneously?

Likewise given you've added the support for computing the vector version in order to make bool work, should we just handle the intrinsics that produce a vector as well?

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Those had no hits so I just didn't want to add more code and tests 🙂 Although, even EQ/NE don't have hits, I just found a use case outside.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think this is the type of scenario where it’s a core comparison on SIMD and so we want to generally handle it even if our own code doesn’t have any hits today.

it’s odd to cover some of the relational comparisons and not the others


#ifdef TARGET_ARM64
case NI_AdvSimd_Add:
case NI_AdvSimd_Arm64_Add:
Expand DownExpand Up@@ -10269,14 +10347,35 @@ void Compiler::fgValueNumberSsaVarDef(GenTreeLclVarCommon* lcl)
static bool GetStaticFieldSeqAndAddress(ValueNumStore* vnStore, GenTree* tree, ssize_t* byteOffset, FieldSeq** pFseq)
{
VNFuncApp funcApp;
if (vnStore->GetVNFunc(tree->gtVNPair.GetLiberal(), &funcApp) && (funcApp.m_func == VNF_PtrToStatic))
if (vnStore->GetVNFunc(tree->gtVNPair.GetLiberal(), &funcApp))
{
FieldSeq* fseq = vnStore->FieldSeqVNToFieldSeq(funcApp.m_args[1]);
if (fseq->GetKind() == FieldSeq::FieldKind::SimpleStatic)
if (funcApp.m_func == VNF_PtrToStatic)
{
*byteOffset = vnStore->ConstantValue<ssize_t>(funcApp.m_args[2]);
*pFseq = fseq;
return true;
FieldSeq* fseq = vnStore->FieldSeqVNToFieldSeq(funcApp.m_args[1]);
if (fseq->GetKind() == FieldSeq::FieldKind::SimpleStatic)
{
*byteOffset = vnStore->ConstantValue<ssize_t>(funcApp.m_args[2]);
*pFseq = fseq;
return true;
}
}
else if (funcApp.m_func == VNFunc(GT_ADD))
{
// Handle ADD(STATIC_HDL, OFFSET) via VN (the logic in this method mostly works with plain tree nodes)
if (vnStore->IsVNHandle(funcApp.m_args[0]) &&
(vnStore->GetHandleFlags(funcApp.m_args[0]) == GTF_ICON_STATIC_HDL) &&
vnStore->IsVNConstant(funcApp.m_args[1]) && !vnStore->IsVNHandle(funcApp.m_args[1]))
{
FieldSeq* fldSeq = vnStore->GetFieldSeqFromAddress(funcApp.m_args[0]);
if (fldSeq != nullptr)
{
assert(fldSeq->GetKind() == FieldSeq::FieldKind::SimpleStaticKnownAddress);
*pFseq = fldSeq;
*byteOffset = vnStore->CoercedConstantValue<ssize_t>(funcApp.m_args[0]) - fldSeq->GetOffset() +
vnStore->CoercedConstantValue<ssize_t>(funcApp.m_args[1]);
return true;
}
}
}
}
ssize_t val = 0;
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -733,4 +733,34 @@ public static void XorTests()
^ Vector128.Create((double)(+1), +1)
);
}

[Fact]
public static void EqualityTests()
{
Assert.True(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0) ==
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.False(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0) !=
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.False(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xFF) ==
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.True(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xFF) !=
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.True(
Vector128.EqualsAll(
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8),
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8)));
Assert.False(
Vector128.EqualsAll(
Vector128.Create((ushort)(0x0001), 0xFFFF, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8),
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8)));
Assert.Equal(
Vector128.Create(4294967295, 0, 4294967295, 4294967295),
Vector128.Equals(
Vector128.Create((uint)(0x0000_0001), 0xFFFF_FFFE, 0x0000_0003, 0xFFFF_FFFC),
Vector128.Create((uint)(0x0000_0001), 0x0000_0001, 0x0000_0003, 0xFFFF_FFFC)));
}
}
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
26 changes: 26 additions & 0 deletions src/coreclr/jit/simd.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -440,6 +440,12 @@ TBase EvaluateBinaryScalarRSZ(TBase arg0, TBase arg1)
return arg0 >> (arg1 & ((sizeof(TBase) * 8) - 1));
}

template <typename TBase>
TBase GetAllBitsSetScalar()
{
return ~static_cast<TBase>(0);
}

template <>
inline int8_t EvaluateBinaryScalarRSZ<int8_t>(int8_t arg0, int8_t arg1)
{
Expand DownExpand Up@@ -520,6 +526,26 @@ TBase EvaluateBinaryScalarSpecialized(genTreeOps oper, TBase arg0, TBase arg1)
return arg0 ^ arg1;
}

case GT_EQ:
{
#ifdef _MSC_VER
// Floating point is not supported
assert(&typeid(TBase) != &typeid(float));
assert(&typeid(TBase) != &typeid(double));
#endif // _MSC_VER
return arg0 == arg1 ? GetAllBitsSetScalar<TBase>() : 0;
}

case GT_NE:
{
#ifdef _MSC_VER
// Floating point is not supported
assert(&typeid(TBase) != &typeid(float));
assert(&typeid(TBase) != &typeid(double));
#endif // _MSC_VER
return arg0 != arg1 ? GetAllBitsSetScalar<TBase>() : 0;
}

default:
{
unreached();
Expand Down
111 changes: 105 additions & 6 deletions src/coreclr/jit/valuenum.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -6812,6 +6812,47 @@ ValueNum EvaluateBinarySimd(ValueNumStore* vns,
}
#endif // TARGET_XARCH

case TYP_BOOL:
{
assert((oper == GT_EQ) || (oper == GT_NE));

var_types vn1Type = vns->TypeOfVN(arg0VN);
var_types vn2Type = vns->TypeOfVN(arg1VN);

assert((vn1Type == vn2Type) && varTypeIsSIMD(vn1Type));
assert(!varTypeIsFloating(baseType));

ValueNum packed = EvaluateBinarySimd(vns, GT_EQ, scalar, vn1Type, baseType, arg0VN, arg1VN);

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

shouldn't this be EvaluateBinarySimd(vns, oper, ...)?

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I wonder if it would be simpler to implement this as EvaluateVector and then simplify check result IsAllBitsSet or !IsZero

Which would also simplify the other relational comparisons

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

shouldn't this be EvaluateBinarySimd(vns, oper, ...)?

It makes it harder to check output, doesn't it?
Currently I just pass EQ and then return AllBitsSet or !AllBitsSet depending on source oper

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Does it? We have a simple property for any simd_T and it makes it easier to cover the other cases.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

That is, if you just evaluate per element you get a result simd_T result and can then just check IsAllBitsSet or IsZero

bool allBitsSet = false;
if (vn1Type == TYP_SIMD8)
{
allBitsSet = GetConstantSimd8(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD12)
{
allBitsSet = GetConstantSimd12(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD16)
{
allBitsSet = GetConstantSimd16(vns, baseType, packed).IsAllBitsSet();
}
#if defined(TARGET_XARCH)
else if (vn1Type == TYP_SIMD32)
{
allBitsSet = GetConstantSimd32(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD64)
{
allBitsSet = GetConstantSimd64(vns, baseType, packed).IsAllBitsSet();
}
#endif
else
{
unreached();
}
return vns->VNForIntCon(oper == GT_EQ ? allBitsSet : !allBitsSet);
}

default:
{
unreached();
Expand DownExpand Up@@ -7167,6 +7208,43 @@ ValueNum ValueNumStore::EvalHWIntrinsicFunBinary(var_types type,

switch (ni)
{
#ifdef TARGET_ARM64
case NI_Vector64_op_Equality:
case NI_Vector128_op_Equality:
case NI_Vector64_EqualsAll:
case NI_Vector128_EqualsAll:
#else
case NI_Vector128_op_Equality:
case NI_Vector256_op_Equality:
case NI_Vector512_op_Equality:
case NI_Vector128_EqualsAll:
case NI_Vector256_EqualsAll:
case NI_Vector512_EqualsAll:
#endif
{
if (!varTypeIsFloating(baseType))
{
return EvaluateBinarySimd(this, GT_EQ, /* scalar */ false, type, baseType, arg0VN, arg1VN);
}
break;
}

#ifdef TARGET_ARM64
case NI_Vector64_op_Inequality:
case NI_Vector128_op_Inequality:
#else
case NI_Vector128_op_Inequality:
case NI_Vector256_op_Inequality:
case NI_Vector512_op_Inequality:
#endif
{
if (!varTypeIsFloating(baseType))
{
return EvaluateBinarySimd(this, GT_NE, /* scalar */ false, type, baseType, arg0VN, arg1VN);
}
break;
}

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Do we want to handle GT, LT, GE, and LE simultaneously?

Likewise given you've added the support for computing the vector version in order to make bool work, should we just handle the intrinsics that produce a vector as well?

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Those had no hits so I just didn't want to add more code and tests 🙂 Although, even EQ/NE don't have hits, I just found a use case outside.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think this is the type of scenario where it’s a core comparison on SIMD and so we want to generally handle it even if our own code doesn’t have any hits today.

it’s odd to cover some of the relational comparisons and not the others


#ifdef TARGET_ARM64
case NI_AdvSimd_Add:
case NI_AdvSimd_Arm64_Add:
Expand DownExpand Up@@ -10269,14 +10347,35 @@ void Compiler::fgValueNumberSsaVarDef(GenTreeLclVarCommon* lcl)
static bool GetStaticFieldSeqAndAddress(ValueNumStore* vnStore, GenTree* tree, ssize_t* byteOffset, FieldSeq** pFseq)
{
VNFuncApp funcApp;
if (vnStore->GetVNFunc(tree->gtVNPair.GetLiberal(), &funcApp) && (funcApp.m_func == VNF_PtrToStatic))
if (vnStore->GetVNFunc(tree->gtVNPair.GetLiberal(), &funcApp))
{
FieldSeq* fseq = vnStore->FieldSeqVNToFieldSeq(funcApp.m_args[1]);
if (fseq->GetKind() == FieldSeq::FieldKind::SimpleStatic)
if (funcApp.m_func == VNF_PtrToStatic)
{
*byteOffset = vnStore->ConstantValue<ssize_t>(funcApp.m_args[2]);
*pFseq = fseq;
return true;
FieldSeq* fseq = vnStore->FieldSeqVNToFieldSeq(funcApp.m_args[1]);
if (fseq->GetKind() == FieldSeq::FieldKind::SimpleStatic)
{
*byteOffset = vnStore->ConstantValue<ssize_t>(funcApp.m_args[2]);
*pFseq = fseq;
return true;
}
}
else if (funcApp.m_func == VNFunc(GT_ADD))
{
// Handle ADD(STATIC_HDL, OFFSET) via VN (the logic in this method mostly works with plain tree nodes)
if (vnStore->IsVNHandle(funcApp.m_args[0]) &&
(vnStore->GetHandleFlags(funcApp.m_args[0]) == GTF_ICON_STATIC_HDL) &&
vnStore->IsVNConstant(funcApp.m_args[1]) && !vnStore->IsVNHandle(funcApp.m_args[1]))
{
FieldSeq* fldSeq = vnStore->GetFieldSeqFromAddress(funcApp.m_args[0]);
if (fldSeq != nullptr)
{
assert(fldSeq->GetKind() == FieldSeq::FieldKind::SimpleStaticKnownAddress);
*pFseq = fldSeq;
*byteOffset = vnStore->CoercedConstantValue<ssize_t>(funcApp.m_args[0]) - fldSeq->GetOffset() +
vnStore->CoercedConstantValue<ssize_t>(funcApp.m_args[1]);
return true;
}
}
}
}
ssize_t val = 0;
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -733,4 +733,34 @@ public static void XorTests()
^ Vector128.Create((double)(+1), +1)
);
}

[Fact]
public static void EqualityTests()
{
Assert.True(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0) ==
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.False(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0) !=
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.False(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xFF) ==
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.True(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xFF) !=
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.True(
Vector128.EqualsAll(
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8),
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8)));
Assert.False(
Vector128.EqualsAll(
Vector128.Create((ushort)(0x0001), 0xFFFF, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8),
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8)));
Assert.Equal(
Vector128.Create(4294967295, 0, 4294967295, 4294967295),
Vector128.Equals(
Vector128.Create((uint)(0x0000_0001), 0xFFFF_FFFE, 0x0000_0003, 0xFFFF_FFFC),
Vector128.Create((uint)(0x0000_0001), 0x0000_0001, 0x0000_0003, 0xFFFF_FFFC)));
}
}
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
26 changes: 26 additions & 0 deletions src/coreclr/jit/simd.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -440,6 +440,12 @@ TBase EvaluateBinaryScalarRSZ(TBase arg0, TBase arg1)
return arg0 >> (arg1 & ((sizeof(TBase) * 8) - 1));
}

template <typename TBase>
TBase GetAllBitsSetScalar()
{
return ~static_cast<TBase>(0);
}

template <>
inline int8_t EvaluateBinaryScalarRSZ<int8_t>(int8_t arg0, int8_t arg1)
{
Expand DownExpand Up@@ -520,6 +526,26 @@ TBase EvaluateBinaryScalarSpecialized(genTreeOps oper, TBase arg0, TBase arg1)
return arg0 ^ arg1;
}

case GT_EQ:
{
#ifdef _MSC_VER
// Floating point is not supported
assert(&typeid(TBase) != &typeid(float));
assert(&typeid(TBase) != &typeid(double));
#endif // _MSC_VER
return arg0 == arg1 ? GetAllBitsSetScalar<TBase>() : 0;
}

case GT_NE:
{
#ifdef _MSC_VER
// Floating point is not supported
assert(&typeid(TBase) != &typeid(float));
assert(&typeid(TBase) != &typeid(double));
#endif // _MSC_VER
return arg0 != arg1 ? GetAllBitsSetScalar<TBase>() : 0;
}

default:
{
unreached();
Expand Down
111 changes: 105 additions & 6 deletions src/coreclr/jit/valuenum.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -6812,6 +6812,47 @@ ValueNum EvaluateBinarySimd(ValueNumStore* vns,
}
#endif // TARGET_XARCH

case TYP_BOOL:
{
assert((oper == GT_EQ) || (oper == GT_NE));

var_types vn1Type = vns->TypeOfVN(arg0VN);
var_types vn2Type = vns->TypeOfVN(arg1VN);

assert((vn1Type == vn2Type) && varTypeIsSIMD(vn1Type));
assert(!varTypeIsFloating(baseType));

ValueNum packed = EvaluateBinarySimd(vns, GT_EQ, scalar, vn1Type, baseType, arg0VN, arg1VN);

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

shouldn't this be EvaluateBinarySimd(vns, oper, ...)?

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I wonder if it would be simpler to implement this as EvaluateVector and then simplify check result IsAllBitsSet or !IsZero

Which would also simplify the other relational comparisons

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

shouldn't this be EvaluateBinarySimd(vns, oper, ...)?

It makes it harder to check output, doesn't it?
Currently I just pass EQ and then return AllBitsSet or !AllBitsSet depending on source oper

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Does it? We have a simple property for any simd_T and it makes it easier to cover the other cases.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

That is, if you just evaluate per element you get a result simd_T result and can then just check IsAllBitsSet or IsZero

bool allBitsSet = false;
if (vn1Type == TYP_SIMD8)
{
allBitsSet = GetConstantSimd8(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD12)
{
allBitsSet = GetConstantSimd12(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD16)
{
allBitsSet = GetConstantSimd16(vns, baseType, packed).IsAllBitsSet();
}
#if defined(TARGET_XARCH)
else if (vn1Type == TYP_SIMD32)
{
allBitsSet = GetConstantSimd32(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD64)
{
allBitsSet = GetConstantSimd64(vns, baseType, packed).IsAllBitsSet();
}
#endif
else
{
unreached();
}
return vns->VNForIntCon(oper == GT_EQ ? allBitsSet : !allBitsSet);
}

default:
{
unreached();
Expand DownExpand Up@@ -7167,6 +7208,43 @@ ValueNum ValueNumStore::EvalHWIntrinsicFunBinary(var_types type,

switch (ni)
{
#ifdef TARGET_ARM64
case NI_Vector64_op_Equality:
case NI_Vector128_op_Equality:
case NI_Vector64_EqualsAll:
case NI_Vector128_EqualsAll:
#else
case NI_Vector128_op_Equality:
case NI_Vector256_op_Equality:
case NI_Vector512_op_Equality:
case NI_Vector128_EqualsAll:
case NI_Vector256_EqualsAll:
case NI_Vector512_EqualsAll:
#endif
{
if (!varTypeIsFloating(baseType))
{
return EvaluateBinarySimd(this, GT_EQ, /* scalar */ false, type, baseType, arg0VN, arg1VN);
}
break;
}

#ifdef TARGET_ARM64
case NI_Vector64_op_Inequality:
case NI_Vector128_op_Inequality:
#else
case NI_Vector128_op_Inequality:
case NI_Vector256_op_Inequality:
case NI_Vector512_op_Inequality:
#endif
{
if (!varTypeIsFloating(baseType))
{
return EvaluateBinarySimd(this, GT_NE, /* scalar */ false, type, baseType, arg0VN, arg1VN);
}
break;
}

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Do we want to handle GT, LT, GE, and LE simultaneously?

Likewise given you've added the support for computing the vector version in order to make bool work, should we just handle the intrinsics that produce a vector as well?

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Those had no hits so I just didn't want to add more code and tests 🙂 Although, even EQ/NE don't have hits, I just found a use case outside.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think this is the type of scenario where it’s a core comparison on SIMD and so we want to generally handle it even if our own code doesn’t have any hits today.

it’s odd to cover some of the relational comparisons and not the others


#ifdef TARGET_ARM64
case NI_AdvSimd_Add:
case NI_AdvSimd_Arm64_Add:
Expand DownExpand Up@@ -10269,14 +10347,35 @@ void Compiler::fgValueNumberSsaVarDef(GenTreeLclVarCommon* lcl)
static bool GetStaticFieldSeqAndAddress(ValueNumStore* vnStore, GenTree* tree, ssize_t* byteOffset, FieldSeq** pFseq)
{
VNFuncApp funcApp;
if (vnStore->GetVNFunc(tree->gtVNPair.GetLiberal(), &funcApp) && (funcApp.m_func == VNF_PtrToStatic))
if (vnStore->GetVNFunc(tree->gtVNPair.GetLiberal(), &funcApp))
{
FieldSeq* fseq = vnStore->FieldSeqVNToFieldSeq(funcApp.m_args[1]);
if (fseq->GetKind() == FieldSeq::FieldKind::SimpleStatic)
if (funcApp.m_func == VNF_PtrToStatic)
{
*byteOffset = vnStore->ConstantValue<ssize_t>(funcApp.m_args[2]);
*pFseq = fseq;
return true;
FieldSeq* fseq = vnStore->FieldSeqVNToFieldSeq(funcApp.m_args[1]);
if (fseq->GetKind() == FieldSeq::FieldKind::SimpleStatic)
{
*byteOffset = vnStore->ConstantValue<ssize_t>(funcApp.m_args[2]);
*pFseq = fseq;
return true;
}
}
else if (funcApp.m_func == VNFunc(GT_ADD))
{
// Handle ADD(STATIC_HDL, OFFSET) via VN (the logic in this method mostly works with plain tree nodes)
if (vnStore->IsVNHandle(funcApp.m_args[0]) &&
(vnStore->GetHandleFlags(funcApp.m_args[0]) == GTF_ICON_STATIC_HDL) &&
vnStore->IsVNConstant(funcApp.m_args[1]) && !vnStore->IsVNHandle(funcApp.m_args[1]))
{
FieldSeq* fldSeq = vnStore->GetFieldSeqFromAddress(funcApp.m_args[0]);
if (fldSeq != nullptr)
{
assert(fldSeq->GetKind() == FieldSeq::FieldKind::SimpleStaticKnownAddress);
*pFseq = fldSeq;
*byteOffset = vnStore->CoercedConstantValue<ssize_t>(funcApp.m_args[0]) - fldSeq->GetOffset() +
vnStore->CoercedConstantValue<ssize_t>(funcApp.m_args[1]);
return true;
}
}
}
}
ssize_t val = 0;
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -733,4 +733,34 @@ public static void XorTests()
^ Vector128.Create((double)(+1), +1)
);
}

[Fact]
public static void EqualityTests()
{
Assert.True(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0) ==
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.False(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0) !=
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.False(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xFF) ==
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.True(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xFF) !=
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.True(
Vector128.EqualsAll(
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8),
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8)));
Assert.False(
Vector128.EqualsAll(
Vector128.Create((ushort)(0x0001), 0xFFFF, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8),
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8)));
Assert.Equal(
Vector128.Create(4294967295, 0, 4294967295, 4294967295),
Vector128.Equals(
Vector128.Create((uint)(0x0000_0001), 0xFFFF_FFFE, 0x0000_0003, 0xFFFF_FFFC),
Vector128.Create((uint)(0x0000_0001), 0x0000_0001, 0x0000_0003, 0xFFFF_FFFC)));
}
}
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
26 changes: 26 additions & 0 deletions src/coreclr/jit/simd.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -440,6 +440,12 @@ TBase EvaluateBinaryScalarRSZ(TBase arg0, TBase arg1)
return arg0 >> (arg1 & ((sizeof(TBase) * 8) - 1));
}

template <typename TBase>
TBase GetAllBitsSetScalar()
{
return ~static_cast<TBase>(0);
}

template <>
inline int8_t EvaluateBinaryScalarRSZ<int8_t>(int8_t arg0, int8_t arg1)
{
Expand DownExpand Up@@ -520,6 +526,26 @@ TBase EvaluateBinaryScalarSpecialized(genTreeOps oper, TBase arg0, TBase arg1)
return arg0 ^ arg1;
}

case GT_EQ:
{
#ifdef _MSC_VER
// Floating point is not supported
assert(&typeid(TBase) != &typeid(float));
assert(&typeid(TBase) != &typeid(double));
#endif // _MSC_VER
return arg0 == arg1 ? GetAllBitsSetScalar<TBase>() : 0;
}

case GT_NE:
{
#ifdef _MSC_VER
// Floating point is not supported
assert(&typeid(TBase) != &typeid(float));
assert(&typeid(TBase) != &typeid(double));
#endif // _MSC_VER
return arg0 != arg1 ? GetAllBitsSetScalar<TBase>() : 0;
}

default:
{
unreached();
Expand Down
111 changes: 105 additions & 6 deletions src/coreclr/jit/valuenum.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -6812,6 +6812,47 @@ ValueNum EvaluateBinarySimd(ValueNumStore* vns,
}
#endif // TARGET_XARCH

case TYP_BOOL:
{
assert((oper == GT_EQ) || (oper == GT_NE));

var_types vn1Type = vns->TypeOfVN(arg0VN);
var_types vn2Type = vns->TypeOfVN(arg1VN);

assert((vn1Type == vn2Type) && varTypeIsSIMD(vn1Type));
assert(!varTypeIsFloating(baseType));

ValueNum packed = EvaluateBinarySimd(vns, GT_EQ, scalar, vn1Type, baseType, arg0VN, arg1VN);

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

shouldn't this be EvaluateBinarySimd(vns, oper, ...)?

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I wonder if it would be simpler to implement this as EvaluateVector and then simplify check result IsAllBitsSet or !IsZero

Which would also simplify the other relational comparisons

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

shouldn't this be EvaluateBinarySimd(vns, oper, ...)?

It makes it harder to check output, doesn't it?
Currently I just pass EQ and then return AllBitsSet or !AllBitsSet depending on source oper

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Does it? We have a simple property for any simd_T and it makes it easier to cover the other cases.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

That is, if you just evaluate per element you get a result simd_T result and can then just check IsAllBitsSet or IsZero

bool allBitsSet = false;
if (vn1Type == TYP_SIMD8)
{
allBitsSet = GetConstantSimd8(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD12)
{
allBitsSet = GetConstantSimd12(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD16)
{
allBitsSet = GetConstantSimd16(vns, baseType, packed).IsAllBitsSet();
}
#if defined(TARGET_XARCH)
else if (vn1Type == TYP_SIMD32)
{
allBitsSet = GetConstantSimd32(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD64)
{
allBitsSet = GetConstantSimd64(vns, baseType, packed).IsAllBitsSet();
}
#endif
else
{
unreached();
}
return vns->VNForIntCon(oper == GT_EQ ? allBitsSet : !allBitsSet);
}

default:
{
unreached();
Expand DownExpand Up@@ -7167,6 +7208,43 @@ ValueNum ValueNumStore::EvalHWIntrinsicFunBinary(var_types type,

switch (ni)
{
#ifdef TARGET_ARM64
case NI_Vector64_op_Equality:
case NI_Vector128_op_Equality:
case NI_Vector64_EqualsAll:
case NI_Vector128_EqualsAll:
#else
case NI_Vector128_op_Equality:
case NI_Vector256_op_Equality:
case NI_Vector512_op_Equality:
case NI_Vector128_EqualsAll:
case NI_Vector256_EqualsAll:
case NI_Vector512_EqualsAll:
#endif
{
if (!varTypeIsFloating(baseType))
{
return EvaluateBinarySimd(this, GT_EQ, /* scalar */ false, type, baseType, arg0VN, arg1VN);
}
break;
}

#ifdef TARGET_ARM64
case NI_Vector64_op_Inequality:
case NI_Vector128_op_Inequality:
#else
case NI_Vector128_op_Inequality:
case NI_Vector256_op_Inequality:
case NI_Vector512_op_Inequality:
#endif
{
if (!varTypeIsFloating(baseType))
{
return EvaluateBinarySimd(this, GT_NE, /* scalar */ false, type, baseType, arg0VN, arg1VN);
}
break;
}

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Do we want to handle GT, LT, GE, and LE simultaneously?

Likewise given you've added the support for computing the vector version in order to make bool work, should we just handle the intrinsics that produce a vector as well?

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Those had no hits so I just didn't want to add more code and tests 🙂 Although, even EQ/NE don't have hits, I just found a use case outside.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think this is the type of scenario where it’s a core comparison on SIMD and so we want to generally handle it even if our own code doesn’t have any hits today.

it’s odd to cover some of the relational comparisons and not the others


#ifdef TARGET_ARM64
case NI_AdvSimd_Add:
case NI_AdvSimd_Arm64_Add:
Expand DownExpand Up@@ -10269,14 +10347,35 @@ void Compiler::fgValueNumberSsaVarDef(GenTreeLclVarCommon* lcl)
static bool GetStaticFieldSeqAndAddress(ValueNumStore* vnStore, GenTree* tree, ssize_t* byteOffset, FieldSeq** pFseq)
{
VNFuncApp funcApp;
if (vnStore->GetVNFunc(tree->gtVNPair.GetLiberal(), &funcApp) && (funcApp.m_func == VNF_PtrToStatic))
if (vnStore->GetVNFunc(tree->gtVNPair.GetLiberal(), &funcApp))
{
FieldSeq* fseq = vnStore->FieldSeqVNToFieldSeq(funcApp.m_args[1]);
if (fseq->GetKind() == FieldSeq::FieldKind::SimpleStatic)
if (funcApp.m_func == VNF_PtrToStatic)
{
*byteOffset = vnStore->ConstantValue<ssize_t>(funcApp.m_args[2]);
*pFseq = fseq;
return true;
FieldSeq* fseq = vnStore->FieldSeqVNToFieldSeq(funcApp.m_args[1]);
if (fseq->GetKind() == FieldSeq::FieldKind::SimpleStatic)
{
*byteOffset = vnStore->ConstantValue<ssize_t>(funcApp.m_args[2]);
*pFseq = fseq;
return true;
}
}
else if (funcApp.m_func == VNFunc(GT_ADD))
{
// Handle ADD(STATIC_HDL, OFFSET) via VN (the logic in this method mostly works with plain tree nodes)
if (vnStore->IsVNHandle(funcApp.m_args[0]) &&
(vnStore->GetHandleFlags(funcApp.m_args[0]) == GTF_ICON_STATIC_HDL) &&
vnStore->IsVNConstant(funcApp.m_args[1]) && !vnStore->IsVNHandle(funcApp.m_args[1]))
{
FieldSeq* fldSeq = vnStore->GetFieldSeqFromAddress(funcApp.m_args[0]);
if (fldSeq != nullptr)
{
assert(fldSeq->GetKind() == FieldSeq::FieldKind::SimpleStaticKnownAddress);
*pFseq = fldSeq;
*byteOffset = vnStore->CoercedConstantValue<ssize_t>(funcApp.m_args[0]) - fldSeq->GetOffset() +
vnStore->CoercedConstantValue<ssize_t>(funcApp.m_args[1]);
return true;
}
}
}
}
ssize_t val = 0;
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -733,4 +733,34 @@ public static void XorTests()
^ Vector128.Create((double)(+1), +1)
);
}

[Fact]
public static void EqualityTests()
{
Assert.True(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0) ==
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.False(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0) !=
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.False(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xFF) ==
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.True(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xFF) !=
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.True(
Vector128.EqualsAll(
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8),
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8)));
Assert.False(
Vector128.EqualsAll(
Vector128.Create((ushort)(0x0001), 0xFFFF, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8),
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8)));
Assert.Equal(
Vector128.Create(4294967295, 0, 4294967295, 4294967295),
Vector128.Equals(
Vector128.Create((uint)(0x0000_0001), 0xFFFF_FFFE, 0x0000_0003, 0xFFFF_FFFC),
Vector128.Create((uint)(0x0000_0001), 0x0000_0001, 0x0000_0003, 0xFFFF_FFFC)));
}
}
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
26 changes: 26 additions & 0 deletions src/coreclr/jit/simd.h
Original file line numberDiff line numberDiff line change
Expand Up@@ -440,6 +440,12 @@ TBase EvaluateBinaryScalarRSZ(TBase arg0, TBase arg1)
return arg0 >> (arg1 & ((sizeof(TBase) * 8) - 1));
}

template <typename TBase>
TBase GetAllBitsSetScalar()
{
return ~static_cast<TBase>(0);
}

template <>
inline int8_t EvaluateBinaryScalarRSZ<int8_t>(int8_t arg0, int8_t arg1)
{
Expand DownExpand Up@@ -520,6 +526,26 @@ TBase EvaluateBinaryScalarSpecialized(genTreeOps oper, TBase arg0, TBase arg1)
return arg0 ^ arg1;
}

case GT_EQ:
{
#ifdef _MSC_VER
// Floating point is not supported
assert(&typeid(TBase) != &typeid(float));
assert(&typeid(TBase) != &typeid(double));
#endif // _MSC_VER
return arg0 == arg1 ? GetAllBitsSetScalar<TBase>() : 0;
}

case GT_NE:
{
#ifdef _MSC_VER
// Floating point is not supported
assert(&typeid(TBase) != &typeid(float));
assert(&typeid(TBase) != &typeid(double));
#endif // _MSC_VER
return arg0 != arg1 ? GetAllBitsSetScalar<TBase>() : 0;
}

default:
{
unreached();
Expand Down
111 changes: 105 additions & 6 deletions src/coreclr/jit/valuenum.cpp
Original file line numberDiff line numberDiff line change
Expand Up@@ -6812,6 +6812,47 @@ ValueNum EvaluateBinarySimd(ValueNumStore* vns,
}
#endif // TARGET_XARCH

case TYP_BOOL:
{
assert((oper == GT_EQ) || (oper == GT_NE));

var_types vn1Type = vns->TypeOfVN(arg0VN);
var_types vn2Type = vns->TypeOfVN(arg1VN);

assert((vn1Type == vn2Type) && varTypeIsSIMD(vn1Type));
assert(!varTypeIsFloating(baseType));

ValueNum packed = EvaluateBinarySimd(vns, GT_EQ, scalar, vn1Type, baseType, arg0VN, arg1VN);

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

shouldn't this be EvaluateBinarySimd(vns, oper, ...)?

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I wonder if it would be simpler to implement this as EvaluateVector and then simplify check result IsAllBitsSet or !IsZero

Which would also simplify the other relational comparisons

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

shouldn't this be EvaluateBinarySimd(vns, oper, ...)?

It makes it harder to check output, doesn't it?
Currently I just pass EQ and then return AllBitsSet or !AllBitsSet depending on source oper

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Does it? We have a simple property for any simd_T and it makes it easier to cover the other cases.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

That is, if you just evaluate per element you get a result simd_T result and can then just check IsAllBitsSet or IsZero

bool allBitsSet = false;
if (vn1Type == TYP_SIMD8)
{
allBitsSet = GetConstantSimd8(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD12)
{
allBitsSet = GetConstantSimd12(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD16)
{
allBitsSet = GetConstantSimd16(vns, baseType, packed).IsAllBitsSet();
}
#if defined(TARGET_XARCH)
else if (vn1Type == TYP_SIMD32)
{
allBitsSet = GetConstantSimd32(vns, baseType, packed).IsAllBitsSet();
}
else if (vn1Type == TYP_SIMD64)
{
allBitsSet = GetConstantSimd64(vns, baseType, packed).IsAllBitsSet();
}
#endif
else
{
unreached();
}
return vns->VNForIntCon(oper == GT_EQ ? allBitsSet : !allBitsSet);
}

default:
{
unreached();
Expand DownExpand Up@@ -7167,6 +7208,43 @@ ValueNum ValueNumStore::EvalHWIntrinsicFunBinary(var_types type,

switch (ni)
{
#ifdef TARGET_ARM64
case NI_Vector64_op_Equality:
case NI_Vector128_op_Equality:
case NI_Vector64_EqualsAll:
case NI_Vector128_EqualsAll:
#else
case NI_Vector128_op_Equality:
case NI_Vector256_op_Equality:
case NI_Vector512_op_Equality:
case NI_Vector128_EqualsAll:
case NI_Vector256_EqualsAll:
case NI_Vector512_EqualsAll:
#endif
{
if (!varTypeIsFloating(baseType))
{
return EvaluateBinarySimd(this, GT_EQ, /* scalar */ false, type, baseType, arg0VN, arg1VN);
}
break;
}

#ifdef TARGET_ARM64
case NI_Vector64_op_Inequality:
case NI_Vector128_op_Inequality:
#else
case NI_Vector128_op_Inequality:
case NI_Vector256_op_Inequality:
case NI_Vector512_op_Inequality:
#endif
{
if (!varTypeIsFloating(baseType))
{
return EvaluateBinarySimd(this, GT_NE, /* scalar */ false, type, baseType, arg0VN, arg1VN);
}
break;
}

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Do we want to handle GT, LT, GE, and LE simultaneously?

Likewise given you've added the support for computing the vector version in order to make bool work, should we just handle the intrinsics that produce a vector as well?

Copy link
Copy Markdown
MemberAuthor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Those had no hits so I just didn't want to add more code and tests 🙂 Although, even EQ/NE don't have hits, I just found a use case outside.

Copy link
Copy Markdown
Member

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I think this is the type of scenario where it’s a core comparison on SIMD and so we want to generally handle it even if our own code doesn’t have any hits today.

it’s odd to cover some of the relational comparisons and not the others


#ifdef TARGET_ARM64
case NI_AdvSimd_Add:
case NI_AdvSimd_Arm64_Add:
Expand DownExpand Up@@ -10269,14 +10347,35 @@ void Compiler::fgValueNumberSsaVarDef(GenTreeLclVarCommon* lcl)
static bool GetStaticFieldSeqAndAddress(ValueNumStore* vnStore, GenTree* tree, ssize_t* byteOffset, FieldSeq** pFseq)
{
VNFuncApp funcApp;
if (vnStore->GetVNFunc(tree->gtVNPair.GetLiberal(), &funcApp) && (funcApp.m_func == VNF_PtrToStatic))
if (vnStore->GetVNFunc(tree->gtVNPair.GetLiberal(), &funcApp))
{
FieldSeq* fseq = vnStore->FieldSeqVNToFieldSeq(funcApp.m_args[1]);
if (fseq->GetKind() == FieldSeq::FieldKind::SimpleStatic)
if (funcApp.m_func == VNF_PtrToStatic)
{
*byteOffset = vnStore->ConstantValue<ssize_t>(funcApp.m_args[2]);
*pFseq = fseq;
return true;
FieldSeq* fseq = vnStore->FieldSeqVNToFieldSeq(funcApp.m_args[1]);
if (fseq->GetKind() == FieldSeq::FieldKind::SimpleStatic)
{
*byteOffset = vnStore->ConstantValue<ssize_t>(funcApp.m_args[2]);
*pFseq = fseq;
return true;
}
}
else if (funcApp.m_func == VNFunc(GT_ADD))
{
// Handle ADD(STATIC_HDL, OFFSET) via VN (the logic in this method mostly works with plain tree nodes)
if (vnStore->IsVNHandle(funcApp.m_args[0]) &&
(vnStore->GetHandleFlags(funcApp.m_args[0]) == GTF_ICON_STATIC_HDL) &&
vnStore->IsVNConstant(funcApp.m_args[1]) && !vnStore->IsVNHandle(funcApp.m_args[1]))
{
FieldSeq* fldSeq = vnStore->GetFieldSeqFromAddress(funcApp.m_args[0]);
if (fldSeq != nullptr)
{
assert(fldSeq->GetKind() == FieldSeq::FieldKind::SimpleStaticKnownAddress);
*pFseq = fldSeq;
*byteOffset = vnStore->CoercedConstantValue<ssize_t>(funcApp.m_args[0]) - fldSeq->GetOffset() +
vnStore->CoercedConstantValue<ssize_t>(funcApp.m_args[1]);
return true;
}
}
}
}
ssize_t val = 0;
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -733,4 +733,34 @@ public static void XorTests()
^ Vector128.Create((double)(+1), +1)
);
}

[Fact]
public static void EqualityTests()
{
Assert.True(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0) ==
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.False(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0) !=
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.False(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xFF) ==
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.True(
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xFF) !=
Vector128.Create((byte)(0x01), 0xFE, 0x03, 0xFC, 0x05, 0xFA, 0x07, 0xF8, 0x09, 0xF6, 0x0B, 0xF4, 0x0D, 0xF2, 0x0F, 0xF0));
Assert.True(
Vector128.EqualsAll(
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8),
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8)));
Assert.False(
Vector128.EqualsAll(
Vector128.Create((ushort)(0x0001), 0xFFFF, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8),
Vector128.Create((ushort)(0x0001), 0xFFFE, 0x0003, 0xFFFC, 0x0005, 0xFFFA, 0x0007, 0xFFF8)));
Assert.Equal(
Vector128.Create(4294967295, 0, 4294967295, 4294967295),
Vector128.Equals(
Vector128.Create((uint)(0x0000_0001), 0xFFFF_FFFE, 0x0000_0003, 0xFFFF_FFFC),
Vector128.Create((uint)(0x0000_0001), 0x0000_0001, 0x0000_0003, 0xFFFF_FFFC)));
}
}