Skip to content

[Arm64] addressing mode inefficiencies in Guid:op_Equality(Guid,Guid):bool #35622

Description

@BruceForstall

The arm64 generated code for Guid::op_Equality() could be better by (1) incorporating the fp address calculation into the ldr addressing modes, and (2) not using stack at all.

The code:

publicstaticbooloperator==(Guida,Guidb)=>a._a==b._a&&Unsafe.Add(refa._a,1)==Unsafe.Add(refb._a,1)&&Unsafe.Add(refa._a,2)==Unsafe.Add(refb._a,2)&&Unsafe.Add(refa._a,3)==Unsafe.Add(refb._a,3);

This code itself is weird, comparing 4 int values instead of comparing field-by-field of one int, two short, and eight byte. It should compare 2 long on 64-bit.

x64 code is pretty direct translation of this C# code.

arm64 first pushes the 2 16-byte struct-in-register-pair arguments to stack, then reloads each 4-byte element one at a time to compare. The base address of the stack locals are computed over and over, instead of being folded into the subsequent addressing modes that add the offset.

x64 assembly
G_M24558_IG01: ;; bbWeight=1 PerfScore 0.00G_M24558_IG02: 8B01 moveax, dword ptr [rcx] 3B02 cmpeax, dword ptr [rdx] 751D jne SHORT G_M24558_IG05 ;; bbWeight=1 PerfScore 5.00G_M24558_IG03: 8B4104 moveax, dword ptr [rcx+4] 3B4204 cmpeax, dword ptr [rdx+4]7515jne SHORT G_M24558_IG05 8B4108 moveax, dword ptr [rcx+8] 3B4208 cmpeax, dword ptr [rdx+8] 750D jne SHORT G_M24558_IG05 8B410C moveax, dword ptr [rcx+12] 3B420C cmpeax, dword ptr [rdx+12] 0F94C0 sete al 0FB6C0 movzxrax,al ;; bbWeight=0.50 PerfScore 7.63G_M24558_IG04: C3 ret ;; bbWeight=0.50 PerfScore 0.50G_M24558_IG05: 33C0 xoreax,eax ;; bbWeight=0.50 PerfScore 0.13G_M24558_IG06: C3 ret
arm64 assembly
G_M24558_IG01: A9BD7BFD stp fp, lr,[sp,#-48]! 910003FD mov fp,sp F90013A0 str x0,[fp,#32] F90017A1 str x1,[fp,#40] F9000BA2 str x2,[fp,#16] F9000FA3 str x3,[fp,#24] ;; bbWeight=1 PerfScore 5.50G_M24558_IG02: B94023A0 ldr w0,[fp,#32] B94013A1 ldr w1,[fp,#16] 6B01001F cmp w0, w1 540002A1 bne G_M24558_IG05 ;; bbWeight=1 PerfScore 5.50G_M24558_IG03: 910083A0 add x0, fp, #32 B9400400 ldr w0,[x0,#4] 910043A1 add x1, fp, #16 B9400421 ldr w1,[x1,#4] 6B01001F cmp w0, w1 540001E1 bne G_M24558_IG05 910083A0 add x0, fp, #32 B9400800 ldr w0,[x0,#8] 910043A1 add x1, fp, #16 B9400821 ldr w1,[x1,#8] 6B01001F cmp w0, w154000121 bne G_M24558_IG05 910083A0 add x0, fp, #32 B9400C00 ldr w0,[x0,#12] 910043A1 add x1, fp, #16 B9400C21 ldr w1,[x1,#12] 6B01001F cmp w0, w1 9A9F17E0 cset x0, eq ;; bbWeight=0.50 PerfScore 12.50G_M24558_IG04: A8C37BFD ldp fp, lr,[sp],#48 D65F03C0 ret lr ;; bbWeight=0.50 PerfScore 1.00G_M24558_IG05:52800000mov w0, #0 ;; bbWeight=0.50 PerfScore 0.25G_M24558_IG06: A8C37BFD ldp fp, lr,[sp],#48 D65F03C0 ret lr
Possible arm64 assembly after fixing address calculations
G_M24558_IG01: stp fp, lr,[sp,#-48]!mov fp,spstr x0,[fp,#32]str x1,[fp,#40]str x2,[fp,#16]str x3,[fp,#24]G_M24558_IG02: ldr w0,[fp,#32] ldr w1,[fp,#16]cmp w0, w1 bne G_M24558_IG05G_M24558_IG03: ldr w0,[fp,#36] ldr w1,[fp,#20]cmp w0, w1 bne G_M24558_IG05 ldr w0,[fp,#40] ldr w1,[fp,#24]cmp w0, w1 bne G_M24558_IG05 ldr w0,[fp,#44] ldr w1,[fp,#28]cmp w0, w1 cset x0, eqG_M24558_IG04: ldp fp, lr,[sp],#48ret lrG_M24558_IG05:mov w0, #0G_M24558_IG06: ldp fp, lr,[sp],#48ret lr

The JIT shouldn't need to put the argument structs on the stack at all. In which case we could generate code like the following (also assuming we can compare full registers, and not 4 bytes at a time).

Possible arm64 assembly fully optimized
G_M24558_IG01: stp fp, lr,[sp,#-16]!mov fp,spG_M24558_IG02:cmp x0, x2 bne G_M24558_IG05cmp x1, x3 cset x0, eqG_M24558_IG04: ldp fp, lr,[sp],#16ret lrG_M24558_IG05:mov w0, #0G_M24558_IG06: ldp fp, lr,[sp],#16ret lr

category:cq
theme:optimization
skill-level:intermediate
cost:medium

Metadata

Metadata

Assignees

Labels

Type

No type

Projects

No projects

    Relationships

    None yet

    Development

    No branches or pull requests

    Issue actions

    , 'i'); if (__m === '*' || __re.test(location.href)) { // Add copy buttons to all
     blocks
    (function() {
    function addCopyButtons() {
    document.querySelectorAll('pre code').forEach(function(codeBlock) {
    if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;
    codeBlock.parentElement.setAttribute('data-copy-added', 'true');
    var btn = document.createElement('button');
    btn.textContent = 'Copy';
    btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';
    btn.onmouseover = function() { this.style.opacity = '1'; };
    btn.onmouseout = function() { this.style.opacity = '0.7'; };
    btn.onclick = function() {
    navigator.clipboard.writeText(codeBlock.textContent).then(function() {
    btn.textContent = 'Copied!';
    setTimeout(function() { btn.textContent = 'Copy'; }, 1500);
    });
    };
    codeBlock.parentElement.style.position = 'relative';
    codeBlock.parentElement.appendChild(btn);
    });
    }
    addCopyButtons();
    // Re-run on dynamic content
    var observer = new MutationObserver(addCopyButtons);
    observer.observe(document.body, { childList: true, subtree: true });
    })();
    }
    } catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
    })();
    (function(){
    try {
    var __m = "github.com";
    var __re = new RegExp('^' + "github\\.com" + '
    [Arm64] addressing mode inefficiencies in Guid:op_Equality(Guid,Guid):bool · Issue #35622 · dotnet/runtime · GitHub
    Skip to content

    [Arm64] addressing mode inefficiencies in Guid:op_Equality(Guid,Guid):bool #35622

    Description

    @BruceForstall

    The arm64 generated code for Guid::op_Equality() could be better by (1) incorporating the fp address calculation into the ldr addressing modes, and (2) not using stack at all.

    The code:

    publicstaticbooloperator==(Guida,Guidb)=>a._a==b._a&&Unsafe.Add(refa._a,1)==Unsafe.Add(refb._a,1)&&Unsafe.Add(refa._a,2)==Unsafe.Add(refb._a,2)&&Unsafe.Add(refa._a,3)==Unsafe.Add(refb._a,3);

    This code itself is weird, comparing 4 int values instead of comparing field-by-field of one int, two short, and eight byte. It should compare 2 long on 64-bit.

    x64 code is pretty direct translation of this C# code.

    arm64 first pushes the 2 16-byte struct-in-register-pair arguments to stack, then reloads each 4-byte element one at a time to compare. The base address of the stack locals are computed over and over, instead of being folded into the subsequent addressing modes that add the offset.

    x64 assembly
    G_M24558_IG01: ;; bbWeight=1 PerfScore 0.00G_M24558_IG02: 8B01 moveax, dword ptr [rcx] 3B02 cmpeax, dword ptr [rdx] 751D jne SHORT G_M24558_IG05 ;; bbWeight=1 PerfScore 5.00G_M24558_IG03: 8B4104 moveax, dword ptr [rcx+4] 3B4204 cmpeax, dword ptr [rdx+4]7515jne SHORT G_M24558_IG05 8B4108 moveax, dword ptr [rcx+8] 3B4208 cmpeax, dword ptr [rdx+8] 750D jne SHORT G_M24558_IG05 8B410C moveax, dword ptr [rcx+12] 3B420C cmpeax, dword ptr [rdx+12] 0F94C0 sete al 0FB6C0 movzxrax,al ;; bbWeight=0.50 PerfScore 7.63G_M24558_IG04: C3 ret ;; bbWeight=0.50 PerfScore 0.50G_M24558_IG05: 33C0 xoreax,eax ;; bbWeight=0.50 PerfScore 0.13G_M24558_IG06: C3 ret
    arm64 assembly
    G_M24558_IG01: A9BD7BFD stp fp, lr,[sp,#-48]! 910003FD mov fp,sp F90013A0 str x0,[fp,#32] F90017A1 str x1,[fp,#40] F9000BA2 str x2,[fp,#16] F9000FA3 str x3,[fp,#24] ;; bbWeight=1 PerfScore 5.50G_M24558_IG02: B94023A0 ldr w0,[fp,#32] B94013A1 ldr w1,[fp,#16] 6B01001F cmp w0, w1 540002A1 bne G_M24558_IG05 ;; bbWeight=1 PerfScore 5.50G_M24558_IG03: 910083A0 add x0, fp, #32 B9400400 ldr w0,[x0,#4] 910043A1 add x1, fp, #16 B9400421 ldr w1,[x1,#4] 6B01001F cmp w0, w1 540001E1 bne G_M24558_IG05 910083A0 add x0, fp, #32 B9400800 ldr w0,[x0,#8] 910043A1 add x1, fp, #16 B9400821 ldr w1,[x1,#8] 6B01001F cmp w0, w154000121 bne G_M24558_IG05 910083A0 add x0, fp, #32 B9400C00 ldr w0,[x0,#12] 910043A1 add x1, fp, #16 B9400C21 ldr w1,[x1,#12] 6B01001F cmp w0, w1 9A9F17E0 cset x0, eq ;; bbWeight=0.50 PerfScore 12.50G_M24558_IG04: A8C37BFD ldp fp, lr,[sp],#48 D65F03C0 ret lr ;; bbWeight=0.50 PerfScore 1.00G_M24558_IG05:52800000mov w0, #0 ;; bbWeight=0.50 PerfScore 0.25G_M24558_IG06: A8C37BFD ldp fp, lr,[sp],#48 D65F03C0 ret lr
    Possible arm64 assembly after fixing address calculations
    G_M24558_IG01: stp fp, lr,[sp,#-48]!mov fp,spstr x0,[fp,#32]str x1,[fp,#40]str x2,[fp,#16]str x3,[fp,#24]G_M24558_IG02: ldr w0,[fp,#32] ldr w1,[fp,#16]cmp w0, w1 bne G_M24558_IG05G_M24558_IG03: ldr w0,[fp,#36] ldr w1,[fp,#20]cmp w0, w1 bne G_M24558_IG05 ldr w0,[fp,#40] ldr w1,[fp,#24]cmp w0, w1 bne G_M24558_IG05 ldr w0,[fp,#44] ldr w1,[fp,#28]cmp w0, w1 cset x0, eqG_M24558_IG04: ldp fp, lr,[sp],#48ret lrG_M24558_IG05:mov w0, #0G_M24558_IG06: ldp fp, lr,[sp],#48ret lr

    The JIT shouldn't need to put the argument structs on the stack at all. In which case we could generate code like the following (also assuming we can compare full registers, and not 4 bytes at a time).

    Possible arm64 assembly fully optimized
    G_M24558_IG01: stp fp, lr,[sp,#-16]!mov fp,spG_M24558_IG02:cmp x0, x2 bne G_M24558_IG05cmp x1, x3 cset x0, eqG_M24558_IG04: ldp fp, lr,[sp],#16ret lrG_M24558_IG05:mov w0, #0G_M24558_IG06: ldp fp, lr,[sp],#16ret lr

    category:cq
    theme:optimization
    skill-level:intermediate
    cost:medium

    Metadata

    Metadata

    Assignees

    Labels

    Type

    No type

    Projects

    No projects

      Relationships

      None yet

      Development

      No branches or pull requests

      Issue actions

      , 'i'); if (__m === '*' || __re.test(location.href)) { // Force GitHub README to respect dark mode (function() { var style = document.createElement('style'); style.textContent = ' .markdown-body { color-scheme: dark light; } .markdown-body pre { background: #161b22 !important; } .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; } .markdown-body table th, .markdown-body table td { border-color: #30363d !important; } .markdown-body img { background: #0d1117; } .markdown-body blockquote { border-left-color: #8b949e; } .markdown-body hr { border-color: #30363d; } '; document.head.appendChild(style); })(); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' [Arm64] addressing mode inefficiencies in Guid:op_Equality(Guid,Guid):bool · Issue #35622 · dotnet/runtime · GitHub
      Skip to content

      [Arm64] addressing mode inefficiencies in Guid:op_Equality(Guid,Guid):bool #35622

      Description

      @BruceForstall

      The arm64 generated code for Guid::op_Equality() could be better by (1) incorporating the fp address calculation into the ldr addressing modes, and (2) not using stack at all.

      The code:

      publicstaticbooloperator==(Guida,Guidb)=>a._a==b._a&&Unsafe.Add(refa._a,1)==Unsafe.Add(refb._a,1)&&Unsafe.Add(refa._a,2)==Unsafe.Add(refb._a,2)&&Unsafe.Add(refa._a,3)==Unsafe.Add(refb._a,3);

      This code itself is weird, comparing 4 int values instead of comparing field-by-field of one int, two short, and eight byte. It should compare 2 long on 64-bit.

      x64 code is pretty direct translation of this C# code.

      arm64 first pushes the 2 16-byte struct-in-register-pair arguments to stack, then reloads each 4-byte element one at a time to compare. The base address of the stack locals are computed over and over, instead of being folded into the subsequent addressing modes that add the offset.

      x64 assembly
      G_M24558_IG01: ;; bbWeight=1 PerfScore 0.00G_M24558_IG02: 8B01 moveax, dword ptr [rcx] 3B02 cmpeax, dword ptr [rdx] 751D jne SHORT G_M24558_IG05 ;; bbWeight=1 PerfScore 5.00G_M24558_IG03: 8B4104 moveax, dword ptr [rcx+4] 3B4204 cmpeax, dword ptr [rdx+4]7515jne SHORT G_M24558_IG05 8B4108 moveax, dword ptr [rcx+8] 3B4208 cmpeax, dword ptr [rdx+8] 750D jne SHORT G_M24558_IG05 8B410C moveax, dword ptr [rcx+12] 3B420C cmpeax, dword ptr [rdx+12] 0F94C0 sete al 0FB6C0 movzxrax,al ;; bbWeight=0.50 PerfScore 7.63G_M24558_IG04: C3 ret ;; bbWeight=0.50 PerfScore 0.50G_M24558_IG05: 33C0 xoreax,eax ;; bbWeight=0.50 PerfScore 0.13G_M24558_IG06: C3 ret
      arm64 assembly
      G_M24558_IG01: A9BD7BFD stp fp, lr,[sp,#-48]! 910003FD mov fp,sp F90013A0 str x0,[fp,#32] F90017A1 str x1,[fp,#40] F9000BA2 str x2,[fp,#16] F9000FA3 str x3,[fp,#24] ;; bbWeight=1 PerfScore 5.50G_M24558_IG02: B94023A0 ldr w0,[fp,#32] B94013A1 ldr w1,[fp,#16] 6B01001F cmp w0, w1 540002A1 bne G_M24558_IG05 ;; bbWeight=1 PerfScore 5.50G_M24558_IG03: 910083A0 add x0, fp, #32 B9400400 ldr w0,[x0,#4] 910043A1 add x1, fp, #16 B9400421 ldr w1,[x1,#4] 6B01001F cmp w0, w1 540001E1 bne G_M24558_IG05 910083A0 add x0, fp, #32 B9400800 ldr w0,[x0,#8] 910043A1 add x1, fp, #16 B9400821 ldr w1,[x1,#8] 6B01001F cmp w0, w154000121 bne G_M24558_IG05 910083A0 add x0, fp, #32 B9400C00 ldr w0,[x0,#12] 910043A1 add x1, fp, #16 B9400C21 ldr w1,[x1,#12] 6B01001F cmp w0, w1 9A9F17E0 cset x0, eq ;; bbWeight=0.50 PerfScore 12.50G_M24558_IG04: A8C37BFD ldp fp, lr,[sp],#48 D65F03C0 ret lr ;; bbWeight=0.50 PerfScore 1.00G_M24558_IG05:52800000mov w0, #0 ;; bbWeight=0.50 PerfScore 0.25G_M24558_IG06: A8C37BFD ldp fp, lr,[sp],#48 D65F03C0 ret lr
      Possible arm64 assembly after fixing address calculations
      G_M24558_IG01: stp fp, lr,[sp,#-48]!mov fp,spstr x0,[fp,#32]str x1,[fp,#40]str x2,[fp,#16]str x3,[fp,#24]G_M24558_IG02: ldr w0,[fp,#32] ldr w1,[fp,#16]cmp w0, w1 bne G_M24558_IG05G_M24558_IG03: ldr w0,[fp,#36] ldr w1,[fp,#20]cmp w0, w1 bne G_M24558_IG05 ldr w0,[fp,#40] ldr w1,[fp,#24]cmp w0, w1 bne G_M24558_IG05 ldr w0,[fp,#44] ldr w1,[fp,#28]cmp w0, w1 cset x0, eqG_M24558_IG04: ldp fp, lr,[sp],#48ret lrG_M24558_IG05:mov w0, #0G_M24558_IG06: ldp fp, lr,[sp],#48ret lr

      The JIT shouldn't need to put the argument structs on the stack at all. In which case we could generate code like the following (also assuming we can compare full registers, and not 4 bytes at a time).

      Possible arm64 assembly fully optimized
      G_M24558_IG01: stp fp, lr,[sp,#-16]!mov fp,spG_M24558_IG02:cmp x0, x2 bne G_M24558_IG05cmp x1, x3 cset x0, eqG_M24558_IG04: ldp fp, lr,[sp],#16ret lrG_M24558_IG05:mov w0, #0G_M24558_IG06: ldp fp, lr,[sp],#16ret lr

      category:cq
      theme:optimization
      skill-level:intermediate
      cost:medium

      Metadata

      Metadata

      Assignees

      Labels

      Type

      No type

      Projects

      No projects

        Relationships

        None yet

        Development

        No branches or pull requests

        Issue actions

        , 'i'); if (__m === '*' || __re.test(location.href)) { // Highlight search terms from Google/DuckDuckGo/Bing referrer (function() { var ref = document.referrer; var terms = []; if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) { var url = new URL(ref); var q = url.searchParams.get('q') || url.searchParams.get('p'); if (q) { terms = q.split(/\s+/).filter(function(t) { return t.length > 2; }); } } if (terms.length === 0) return; var style = document.createElement('style'); style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }'; document.head.appendChild(style); function highlight(node) { if (node.nodeType === 3) { // text node var text = node.textContent; var found = false; terms.forEach(function(term) { var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\]\\]/g, '\\') + ')', 'gi'); if (regex.test(text)) { found = true; var frag = document.createDocumentFragment(); var parts = text.split(regex); parts.forEach(function(part, i) { if (i % 2 === 0) { frag.appendChild(document.createTextNode(part)); } else { var span = document.createElement('span'); span.className = 'userscript-highlight'; span.textContent = part; frag.appendChild(span); } }); node.parentNode.replaceChild(frag, node); } }); } else if (node.nodeType === 1 && node.childNodes) { // element var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT']; if (!skipTags.includes(node.tagName)) { Array.from(node.childNodes).forEach(highlight); } } } highlight(document.body); // Re-highlight on dynamic content var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1 || node.nodeType === 3) highlight(node); }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' [Arm64] addressing mode inefficiencies in Guid:op_Equality(Guid,Guid):bool · Issue #35622 · dotnet/runtime · GitHub
        Skip to content

        [Arm64] addressing mode inefficiencies in Guid:op_Equality(Guid,Guid):bool #35622

        Description

        @BruceForstall

        The arm64 generated code for Guid::op_Equality() could be better by (1) incorporating the fp address calculation into the ldr addressing modes, and (2) not using stack at all.

        The code:

        publicstaticbooloperator==(Guida,Guidb)=>a._a==b._a&&Unsafe.Add(refa._a,1)==Unsafe.Add(refb._a,1)&&Unsafe.Add(refa._a,2)==Unsafe.Add(refb._a,2)&&Unsafe.Add(refa._a,3)==Unsafe.Add(refb._a,3);

        This code itself is weird, comparing 4 int values instead of comparing field-by-field of one int, two short, and eight byte. It should compare 2 long on 64-bit.

        x64 code is pretty direct translation of this C# code.

        arm64 first pushes the 2 16-byte struct-in-register-pair arguments to stack, then reloads each 4-byte element one at a time to compare. The base address of the stack locals are computed over and over, instead of being folded into the subsequent addressing modes that add the offset.

        x64 assembly
        G_M24558_IG01: ;; bbWeight=1 PerfScore 0.00G_M24558_IG02: 8B01 moveax, dword ptr [rcx] 3B02 cmpeax, dword ptr [rdx] 751D jne SHORT G_M24558_IG05 ;; bbWeight=1 PerfScore 5.00G_M24558_IG03: 8B4104 moveax, dword ptr [rcx+4] 3B4204 cmpeax, dword ptr [rdx+4]7515jne SHORT G_M24558_IG05 8B4108 moveax, dword ptr [rcx+8] 3B4208 cmpeax, dword ptr [rdx+8] 750D jne SHORT G_M24558_IG05 8B410C moveax, dword ptr [rcx+12] 3B420C cmpeax, dword ptr [rdx+12] 0F94C0 sete al 0FB6C0 movzxrax,al ;; bbWeight=0.50 PerfScore 7.63G_M24558_IG04: C3 ret ;; bbWeight=0.50 PerfScore 0.50G_M24558_IG05: 33C0 xoreax,eax ;; bbWeight=0.50 PerfScore 0.13G_M24558_IG06: C3 ret
        arm64 assembly
        G_M24558_IG01: A9BD7BFD stp fp, lr,[sp,#-48]! 910003FD mov fp,sp F90013A0 str x0,[fp,#32] F90017A1 str x1,[fp,#40] F9000BA2 str x2,[fp,#16] F9000FA3 str x3,[fp,#24] ;; bbWeight=1 PerfScore 5.50G_M24558_IG02: B94023A0 ldr w0,[fp,#32] B94013A1 ldr w1,[fp,#16] 6B01001F cmp w0, w1 540002A1 bne G_M24558_IG05 ;; bbWeight=1 PerfScore 5.50G_M24558_IG03: 910083A0 add x0, fp, #32 B9400400 ldr w0,[x0,#4] 910043A1 add x1, fp, #16 B9400421 ldr w1,[x1,#4] 6B01001F cmp w0, w1 540001E1 bne G_M24558_IG05 910083A0 add x0, fp, #32 B9400800 ldr w0,[x0,#8] 910043A1 add x1, fp, #16 B9400821 ldr w1,[x1,#8] 6B01001F cmp w0, w154000121 bne G_M24558_IG05 910083A0 add x0, fp, #32 B9400C00 ldr w0,[x0,#12] 910043A1 add x1, fp, #16 B9400C21 ldr w1,[x1,#12] 6B01001F cmp w0, w1 9A9F17E0 cset x0, eq ;; bbWeight=0.50 PerfScore 12.50G_M24558_IG04: A8C37BFD ldp fp, lr,[sp],#48 D65F03C0 ret lr ;; bbWeight=0.50 PerfScore 1.00G_M24558_IG05:52800000mov w0, #0 ;; bbWeight=0.50 PerfScore 0.25G_M24558_IG06: A8C37BFD ldp fp, lr,[sp],#48 D65F03C0 ret lr
        Possible arm64 assembly after fixing address calculations
        G_M24558_IG01: stp fp, lr,[sp,#-48]!mov fp,spstr x0,[fp,#32]str x1,[fp,#40]str x2,[fp,#16]str x3,[fp,#24]G_M24558_IG02: ldr w0,[fp,#32] ldr w1,[fp,#16]cmp w0, w1 bne G_M24558_IG05G_M24558_IG03: ldr w0,[fp,#36] ldr w1,[fp,#20]cmp w0, w1 bne G_M24558_IG05 ldr w0,[fp,#40] ldr w1,[fp,#24]cmp w0, w1 bne G_M24558_IG05 ldr w0,[fp,#44] ldr w1,[fp,#28]cmp w0, w1 cset x0, eqG_M24558_IG04: ldp fp, lr,[sp],#48ret lrG_M24558_IG05:mov w0, #0G_M24558_IG06: ldp fp, lr,[sp],#48ret lr

        The JIT shouldn't need to put the argument structs on the stack at all. In which case we could generate code like the following (also assuming we can compare full registers, and not 4 bytes at a time).

        Possible arm64 assembly fully optimized
        G_M24558_IG01: stp fp, lr,[sp,#-16]!mov fp,spG_M24558_IG02:cmp x0, x2 bne G_M24558_IG05cmp x1, x3 cset x0, eqG_M24558_IG04: ldp fp, lr,[sp],#16ret lrG_M24558_IG05:mov w0, #0G_M24558_IG06: ldp fp, lr,[sp],#16ret lr

        category:cq
        theme:optimization
        skill-level:intermediate
        cost:medium

        Metadata

        Metadata

        Assignees

        Labels

        Type

        No type

        Projects

        No projects

          Relationships

          None yet

          Development

          No branches or pull requests

          Issue actions

          , 'i'); if (__m === '*' || __re.test(location.href)) { // Strip utm_, fbclid, gclid, etc. from all links on page (function() { var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content', 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid', 'ref', 'ref_src', 'source', 'medium', 'campaign']; function cleanUrl(url) { try { var u = new URL(url, window.location.origin); var changed = false; trackingParams.forEach(function(p) { if (u.searchParams.has(p)) { u.searchParams.delete(p); changed = true; } }); return changed ? u.toString() : url; } catch (e) { return url; } } function cleanLinks() { document.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } cleanLinks(); var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1) { if (node.tagName === 'A') cleanLinks(); node.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + ' [Arm64] addressing mode inefficiencies in Guid:op_Equality(Guid,Guid):bool · Issue #35622 · dotnet/runtime · GitHub
          Skip to content

          [Arm64] addressing mode inefficiencies in Guid:op_Equality(Guid,Guid):bool #35622

          Description

          @BruceForstall

          The arm64 generated code for Guid::op_Equality() could be better by (1) incorporating the fp address calculation into the ldr addressing modes, and (2) not using stack at all.

          The code:

          publicstaticbooloperator==(Guida,Guidb)=>a._a==b._a&&Unsafe.Add(refa._a,1)==Unsafe.Add(refb._a,1)&&Unsafe.Add(refa._a,2)==Unsafe.Add(refb._a,2)&&Unsafe.Add(refa._a,3)==Unsafe.Add(refb._a,3);

          This code itself is weird, comparing 4 int values instead of comparing field-by-field of one int, two short, and eight byte. It should compare 2 long on 64-bit.

          x64 code is pretty direct translation of this C# code.

          arm64 first pushes the 2 16-byte struct-in-register-pair arguments to stack, then reloads each 4-byte element one at a time to compare. The base address of the stack locals are computed over and over, instead of being folded into the subsequent addressing modes that add the offset.

          x64 assembly
          G_M24558_IG01: ;; bbWeight=1 PerfScore 0.00G_M24558_IG02: 8B01 moveax, dword ptr [rcx] 3B02 cmpeax, dword ptr [rdx] 751D jne SHORT G_M24558_IG05 ;; bbWeight=1 PerfScore 5.00G_M24558_IG03: 8B4104 moveax, dword ptr [rcx+4] 3B4204 cmpeax, dword ptr [rdx+4]7515jne SHORT G_M24558_IG05 8B4108 moveax, dword ptr [rcx+8] 3B4208 cmpeax, dword ptr [rdx+8] 750D jne SHORT G_M24558_IG05 8B410C moveax, dword ptr [rcx+12] 3B420C cmpeax, dword ptr [rdx+12] 0F94C0 sete al 0FB6C0 movzxrax,al ;; bbWeight=0.50 PerfScore 7.63G_M24558_IG04: C3 ret ;; bbWeight=0.50 PerfScore 0.50G_M24558_IG05: 33C0 xoreax,eax ;; bbWeight=0.50 PerfScore 0.13G_M24558_IG06: C3 ret
          arm64 assembly
          G_M24558_IG01: A9BD7BFD stp fp, lr,[sp,#-48]! 910003FD mov fp,sp F90013A0 str x0,[fp,#32] F90017A1 str x1,[fp,#40] F9000BA2 str x2,[fp,#16] F9000FA3 str x3,[fp,#24] ;; bbWeight=1 PerfScore 5.50G_M24558_IG02: B94023A0 ldr w0,[fp,#32] B94013A1 ldr w1,[fp,#16] 6B01001F cmp w0, w1 540002A1 bne G_M24558_IG05 ;; bbWeight=1 PerfScore 5.50G_M24558_IG03: 910083A0 add x0, fp, #32 B9400400 ldr w0,[x0,#4] 910043A1 add x1, fp, #16 B9400421 ldr w1,[x1,#4] 6B01001F cmp w0, w1 540001E1 bne G_M24558_IG05 910083A0 add x0, fp, #32 B9400800 ldr w0,[x0,#8] 910043A1 add x1, fp, #16 B9400821 ldr w1,[x1,#8] 6B01001F cmp w0, w154000121 bne G_M24558_IG05 910083A0 add x0, fp, #32 B9400C00 ldr w0,[x0,#12] 910043A1 add x1, fp, #16 B9400C21 ldr w1,[x1,#12] 6B01001F cmp w0, w1 9A9F17E0 cset x0, eq ;; bbWeight=0.50 PerfScore 12.50G_M24558_IG04: A8C37BFD ldp fp, lr,[sp],#48 D65F03C0 ret lr ;; bbWeight=0.50 PerfScore 1.00G_M24558_IG05:52800000mov w0, #0 ;; bbWeight=0.50 PerfScore 0.25G_M24558_IG06: A8C37BFD ldp fp, lr,[sp],#48 D65F03C0 ret lr
          Possible arm64 assembly after fixing address calculations
          G_M24558_IG01: stp fp, lr,[sp,#-48]!mov fp,spstr x0,[fp,#32]str x1,[fp,#40]str x2,[fp,#16]str x3,[fp,#24]G_M24558_IG02: ldr w0,[fp,#32] ldr w1,[fp,#16]cmp w0, w1 bne G_M24558_IG05G_M24558_IG03: ldr w0,[fp,#36] ldr w1,[fp,#20]cmp w0, w1 bne G_M24558_IG05 ldr w0,[fp,#40] ldr w1,[fp,#24]cmp w0, w1 bne G_M24558_IG05 ldr w0,[fp,#44] ldr w1,[fp,#28]cmp w0, w1 cset x0, eqG_M24558_IG04: ldp fp, lr,[sp],#48ret lrG_M24558_IG05:mov w0, #0G_M24558_IG06: ldp fp, lr,[sp],#48ret lr

          The JIT shouldn't need to put the argument structs on the stack at all. In which case we could generate code like the following (also assuming we can compare full registers, and not 4 bytes at a time).

          Possible arm64 assembly fully optimized
          G_M24558_IG01: stp fp, lr,[sp,#-16]!mov fp,spG_M24558_IG02:cmp x0, x2 bne G_M24558_IG05cmp x1, x3 cset x0, eqG_M24558_IG04: ldp fp, lr,[sp],#16ret lrG_M24558_IG05:mov w0, #0G_M24558_IG06: ldp fp, lr,[sp],#16ret lr

          category:cq
          theme:optimization
          skill-level:intermediate
          cost:medium

          Metadata

          Metadata

          Assignees

          Labels

          Type

          No type

          Projects

          No projects

            Relationships

            None yet

            Development

            No branches or pull requests

            Issue actions

            , 'i'); if (__m === '*' || __re.test(location.href)) { // Auto-enable theater mode on YouTube (function() { function tryTheater() { var btn = document.querySelector('button[aria-label="Theater mode"], ytd-player #player button[title="Theater mode"]'); if (btn && !btn.classList.contains('activated')) { btn.click(); } } // Try immediately tryTheater(); // Try after navigation (SPA) var lastUrl = location.href; setInterval(function() { if (location.href !== lastUrl) { lastUrl = location.href; setTimeout(tryTheater, 500); } }, 1000); // Also try on player load var observer = new MutationObserver(tryTheater); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' [Arm64] addressing mode inefficiencies in Guid:op_Equality(Guid,Guid):bool · Issue #35622 · dotnet/runtime · GitHub
            Skip to content

            [Arm64] addressing mode inefficiencies in Guid:op_Equality(Guid,Guid):bool #35622

            Description

            @BruceForstall

            The arm64 generated code for Guid::op_Equality() could be better by (1) incorporating the fp address calculation into the ldr addressing modes, and (2) not using stack at all.

            The code:

            publicstaticbooloperator==(Guida,Guidb)=>a._a==b._a&&Unsafe.Add(refa._a,1)==Unsafe.Add(refb._a,1)&&Unsafe.Add(refa._a,2)==Unsafe.Add(refb._a,2)&&Unsafe.Add(refa._a,3)==Unsafe.Add(refb._a,3);

            This code itself is weird, comparing 4 int values instead of comparing field-by-field of one int, two short, and eight byte. It should compare 2 long on 64-bit.

            x64 code is pretty direct translation of this C# code.

            arm64 first pushes the 2 16-byte struct-in-register-pair arguments to stack, then reloads each 4-byte element one at a time to compare. The base address of the stack locals are computed over and over, instead of being folded into the subsequent addressing modes that add the offset.

            x64 assembly
            G_M24558_IG01: ;; bbWeight=1 PerfScore 0.00G_M24558_IG02: 8B01 moveax, dword ptr [rcx] 3B02 cmpeax, dword ptr [rdx] 751D jne SHORT G_M24558_IG05 ;; bbWeight=1 PerfScore 5.00G_M24558_IG03: 8B4104 moveax, dword ptr [rcx+4] 3B4204 cmpeax, dword ptr [rdx+4]7515jne SHORT G_M24558_IG05 8B4108 moveax, dword ptr [rcx+8] 3B4208 cmpeax, dword ptr [rdx+8] 750D jne SHORT G_M24558_IG05 8B410C moveax, dword ptr [rcx+12] 3B420C cmpeax, dword ptr [rdx+12] 0F94C0 sete al 0FB6C0 movzxrax,al ;; bbWeight=0.50 PerfScore 7.63G_M24558_IG04: C3 ret ;; bbWeight=0.50 PerfScore 0.50G_M24558_IG05: 33C0 xoreax,eax ;; bbWeight=0.50 PerfScore 0.13G_M24558_IG06: C3 ret
            arm64 assembly
            G_M24558_IG01: A9BD7BFD stp fp, lr,[sp,#-48]! 910003FD mov fp,sp F90013A0 str x0,[fp,#32] F90017A1 str x1,[fp,#40] F9000BA2 str x2,[fp,#16] F9000FA3 str x3,[fp,#24] ;; bbWeight=1 PerfScore 5.50G_M24558_IG02: B94023A0 ldr w0,[fp,#32] B94013A1 ldr w1,[fp,#16] 6B01001F cmp w0, w1 540002A1 bne G_M24558_IG05 ;; bbWeight=1 PerfScore 5.50G_M24558_IG03: 910083A0 add x0, fp, #32 B9400400 ldr w0,[x0,#4] 910043A1 add x1, fp, #16 B9400421 ldr w1,[x1,#4] 6B01001F cmp w0, w1 540001E1 bne G_M24558_IG05 910083A0 add x0, fp, #32 B9400800 ldr w0,[x0,#8] 910043A1 add x1, fp, #16 B9400821 ldr w1,[x1,#8] 6B01001F cmp w0, w154000121 bne G_M24558_IG05 910083A0 add x0, fp, #32 B9400C00 ldr w0,[x0,#12] 910043A1 add x1, fp, #16 B9400C21 ldr w1,[x1,#12] 6B01001F cmp w0, w1 9A9F17E0 cset x0, eq ;; bbWeight=0.50 PerfScore 12.50G_M24558_IG04: A8C37BFD ldp fp, lr,[sp],#48 D65F03C0 ret lr ;; bbWeight=0.50 PerfScore 1.00G_M24558_IG05:52800000mov w0, #0 ;; bbWeight=0.50 PerfScore 0.25G_M24558_IG06: A8C37BFD ldp fp, lr,[sp],#48 D65F03C0 ret lr
            Possible arm64 assembly after fixing address calculations
            G_M24558_IG01: stp fp, lr,[sp,#-48]!mov fp,spstr x0,[fp,#32]str x1,[fp,#40]str x2,[fp,#16]str x3,[fp,#24]G_M24558_IG02: ldr w0,[fp,#32] ldr w1,[fp,#16]cmp w0, w1 bne G_M24558_IG05G_M24558_IG03: ldr w0,[fp,#36] ldr w1,[fp,#20]cmp w0, w1 bne G_M24558_IG05 ldr w0,[fp,#40] ldr w1,[fp,#24]cmp w0, w1 bne G_M24558_IG05 ldr w0,[fp,#44] ldr w1,[fp,#28]cmp w0, w1 cset x0, eqG_M24558_IG04: ldp fp, lr,[sp],#48ret lrG_M24558_IG05:mov w0, #0G_M24558_IG06: ldp fp, lr,[sp],#48ret lr

            The JIT shouldn't need to put the argument structs on the stack at all. In which case we could generate code like the following (also assuming we can compare full registers, and not 4 bytes at a time).

            Possible arm64 assembly fully optimized
            G_M24558_IG01: stp fp, lr,[sp,#-16]!mov fp,spG_M24558_IG02:cmp x0, x2 bne G_M24558_IG05cmp x1, x3 cset x0, eqG_M24558_IG04: ldp fp, lr,[sp],#16ret lrG_M24558_IG05:mov w0, #0G_M24558_IG06: ldp fp, lr,[sp],#16ret lr

            category:cq
            theme:optimization
            skill-level:intermediate
            cost:medium

            Metadata

            Metadata

            Assignees

            Labels

            Type

            No type

            Projects

            No projects

              Relationships

              None yet

              Development

              No branches or pull requests

              Issue actions

              , 'i'); if (__m === '*' || __re.test(location.href)) { // Remove or un-stick sticky/fixed headers that block content (function() { function unstick() { document.querySelectorAll('header, nav, [role="banner"], .header, .navbar, .sticky, .fixed-top, [style*="position: fixed"], [style*="position:sticky"]').forEach(function(el) { if (el.style.position === 'fixed' || el.style.position === 'sticky' || getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') { el.style.position = 'static'; el.style.top = 'auto'; el.style.zIndex = 'auto'; } }); } unstick(); var observer = new MutationObserver(unstick); observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] }); })(); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ' [Arm64] addressing mode inefficiencies in Guid:op_Equality(Guid,Guid):bool · Issue #35622 · dotnet/runtime · GitHub
              Skip to content

              [Arm64] addressing mode inefficiencies in Guid:op_Equality(Guid,Guid):bool #35622

              Description

              @BruceForstall

              The arm64 generated code for Guid::op_Equality() could be better by (1) incorporating the fp address calculation into the ldr addressing modes, and (2) not using stack at all.

              The code:

              publicstaticbooloperator==(Guida,Guidb)=>a._a==b._a&&Unsafe.Add(refa._a,1)==Unsafe.Add(refb._a,1)&&Unsafe.Add(refa._a,2)==Unsafe.Add(refb._a,2)&&Unsafe.Add(refa._a,3)==Unsafe.Add(refb._a,3);

              This code itself is weird, comparing 4 int values instead of comparing field-by-field of one int, two short, and eight byte. It should compare 2 long on 64-bit.

              x64 code is pretty direct translation of this C# code.

              arm64 first pushes the 2 16-byte struct-in-register-pair arguments to stack, then reloads each 4-byte element one at a time to compare. The base address of the stack locals are computed over and over, instead of being folded into the subsequent addressing modes that add the offset.

              x64 assembly
              G_M24558_IG01: ;; bbWeight=1 PerfScore 0.00G_M24558_IG02: 8B01 moveax, dword ptr [rcx] 3B02 cmpeax, dword ptr [rdx] 751D jne SHORT G_M24558_IG05 ;; bbWeight=1 PerfScore 5.00G_M24558_IG03: 8B4104 moveax, dword ptr [rcx+4] 3B4204 cmpeax, dword ptr [rdx+4]7515jne SHORT G_M24558_IG05 8B4108 moveax, dword ptr [rcx+8] 3B4208 cmpeax, dword ptr [rdx+8] 750D jne SHORT G_M24558_IG05 8B410C moveax, dword ptr [rcx+12] 3B420C cmpeax, dword ptr [rdx+12] 0F94C0 sete al 0FB6C0 movzxrax,al ;; bbWeight=0.50 PerfScore 7.63G_M24558_IG04: C3 ret ;; bbWeight=0.50 PerfScore 0.50G_M24558_IG05: 33C0 xoreax,eax ;; bbWeight=0.50 PerfScore 0.13G_M24558_IG06: C3 ret
              arm64 assembly
              G_M24558_IG01: A9BD7BFD stp fp, lr,[sp,#-48]! 910003FD mov fp,sp F90013A0 str x0,[fp,#32] F90017A1 str x1,[fp,#40] F9000BA2 str x2,[fp,#16] F9000FA3 str x3,[fp,#24] ;; bbWeight=1 PerfScore 5.50G_M24558_IG02: B94023A0 ldr w0,[fp,#32] B94013A1 ldr w1,[fp,#16] 6B01001F cmp w0, w1 540002A1 bne G_M24558_IG05 ;; bbWeight=1 PerfScore 5.50G_M24558_IG03: 910083A0 add x0, fp, #32 B9400400 ldr w0,[x0,#4] 910043A1 add x1, fp, #16 B9400421 ldr w1,[x1,#4] 6B01001F cmp w0, w1 540001E1 bne G_M24558_IG05 910083A0 add x0, fp, #32 B9400800 ldr w0,[x0,#8] 910043A1 add x1, fp, #16 B9400821 ldr w1,[x1,#8] 6B01001F cmp w0, w154000121 bne G_M24558_IG05 910083A0 add x0, fp, #32 B9400C00 ldr w0,[x0,#12] 910043A1 add x1, fp, #16 B9400C21 ldr w1,[x1,#12] 6B01001F cmp w0, w1 9A9F17E0 cset x0, eq ;; bbWeight=0.50 PerfScore 12.50G_M24558_IG04: A8C37BFD ldp fp, lr,[sp],#48 D65F03C0 ret lr ;; bbWeight=0.50 PerfScore 1.00G_M24558_IG05:52800000mov w0, #0 ;; bbWeight=0.50 PerfScore 0.25G_M24558_IG06: A8C37BFD ldp fp, lr,[sp],#48 D65F03C0 ret lr
              Possible arm64 assembly after fixing address calculations
              G_M24558_IG01: stp fp, lr,[sp,#-48]!mov fp,spstr x0,[fp,#32]str x1,[fp,#40]str x2,[fp,#16]str x3,[fp,#24]G_M24558_IG02: ldr w0,[fp,#32] ldr w1,[fp,#16]cmp w0, w1 bne G_M24558_IG05G_M24558_IG03: ldr w0,[fp,#36] ldr w1,[fp,#20]cmp w0, w1 bne G_M24558_IG05 ldr w0,[fp,#40] ldr w1,[fp,#24]cmp w0, w1 bne G_M24558_IG05 ldr w0,[fp,#44] ldr w1,[fp,#28]cmp w0, w1 cset x0, eqG_M24558_IG04: ldp fp, lr,[sp],#48ret lrG_M24558_IG05:mov w0, #0G_M24558_IG06: ldp fp, lr,[sp],#48ret lr

              The JIT shouldn't need to put the argument structs on the stack at all. In which case we could generate code like the following (also assuming we can compare full registers, and not 4 bytes at a time).

              Possible arm64 assembly fully optimized
              G_M24558_IG01: stp fp, lr,[sp,#-16]!mov fp,spG_M24558_IG02:cmp x0, x2 bne G_M24558_IG05cmp x1, x3 cset x0, eqG_M24558_IG04: ldp fp, lr,[sp],#16ret lrG_M24558_IG05:mov w0, #0G_M24558_IG06: ldp fp, lr,[sp],#16ret lr

              category:cq
              theme:optimization
              skill-level:intermediate
              cost:medium

              Metadata

              Metadata

              Assignees

              Labels

              Type

              No type

              Projects

              No projects

                Relationships

                None yet

                Development

                No branches or pull requests

                Issue actions

                , 'i'); if (__m === '*' || __re.test(location.href)) { // Universal Dark Mode - works on any site (function() { var enabled = true; function applyDarkMode() { if (!enabled) return; // Create style element if it doesn't exist var style = document.getElementById('universal-dark-mode-style'); if (!style) { style = document.createElement('style'); style.id = 'universal-dark-mode-style'; document.head.appendChild(style); } // Dark mode CSS - inverts colors but preserves images/video style.textContent = ' /* Invert everything except media */ html { filter: invert(1) hue-rotate(180deg) !important; background: #1a1a2e !important; } /* Restore images, videos, iframes, canvas */ img, video, iframe, canvas, svg, picture, [style*="background-image"] { filter: invert(1) hue-rotate(180deg) !important; } /* Preserve specific elements that should not be inverted */ .no-dark-mode, .no-dark-mode *, [data-theme="light"], [data-theme="light"], .ace_editor, .ace_editor *, .CodeMirror, .CodeMirror *, .monaco-editor, .monaco-editor *, .markdown-body pre, .markdown-body pre *, .highlight, .highlight *, pre code, pre code * { filter: none !important; } /* Fix common UI elements */ .modal, .popup, .dropdown-menu, .tooltip, .popover { filter: invert(1) hue-rotate(180deg) !important; background: #2d2d44 !important; border-color: #444 !important; } /* Scrollbars */ ::-webkit-scrollbar { background: #1a1a2e !important; } ::-webkit-scrollbar-thumb { background: #444 !important; } ::-webkit-scrollbar-thumb:hover { background: #555 !important; } /* Selection */ ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; } ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; } '; } function removeDarkMode() { var style = document.getElementById('universal-dark-mode-style'); if (style) style.remove(); } // Toggle with Alt+Shift+D document.addEventListener('keydown', function(e) { if (e.altKey && e.shiftKey && e.key === 'D') { e.preventDefault(); enabled = !enabled; if (enabled) { applyDarkMode(); console.log('[Universal Dark Mode] Enabled'); } else { removeDarkMode(); console.log('[Universal Dark Mode] Disabled'); } } }); // Apply on load applyDarkMode(); // Re-apply on dynamic content var observer = new MutationObserver(function(mutations) { if (enabled && !document.getElementById('universal-dark-mode-style')) { applyDarkMode(); } }); observer.observe(document.head, { childList: true }); console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle'); })(); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })(); [Arm64] addressing mode inefficiencies in Guid:op_Equality(Guid,Guid):bool · Issue #35622 · dotnet/runtime · GitHub
                Skip to content

                [Arm64] addressing mode inefficiencies in Guid:op_Equality(Guid,Guid):bool #35622

                Description

                @BruceForstall

                The arm64 generated code for Guid::op_Equality() could be better by (1) incorporating the fp address calculation into the ldr addressing modes, and (2) not using stack at all.

                The code:

                publicstaticbooloperator==(Guida,Guidb)=>a._a==b._a&&Unsafe.Add(refa._a,1)==Unsafe.Add(refb._a,1)&&Unsafe.Add(refa._a,2)==Unsafe.Add(refb._a,2)&&Unsafe.Add(refa._a,3)==Unsafe.Add(refb._a,3);

                This code itself is weird, comparing 4 int values instead of comparing field-by-field of one int, two short, and eight byte. It should compare 2 long on 64-bit.

                x64 code is pretty direct translation of this C# code.

                arm64 first pushes the 2 16-byte struct-in-register-pair arguments to stack, then reloads each 4-byte element one at a time to compare. The base address of the stack locals are computed over and over, instead of being folded into the subsequent addressing modes that add the offset.

                x64 assembly
                G_M24558_IG01: ;; bbWeight=1 PerfScore 0.00G_M24558_IG02: 8B01 moveax, dword ptr [rcx] 3B02 cmpeax, dword ptr [rdx] 751D jne SHORT G_M24558_IG05 ;; bbWeight=1 PerfScore 5.00G_M24558_IG03: 8B4104 moveax, dword ptr [rcx+4] 3B4204 cmpeax, dword ptr [rdx+4]7515jne SHORT G_M24558_IG05 8B4108 moveax, dword ptr [rcx+8] 3B4208 cmpeax, dword ptr [rdx+8] 750D jne SHORT G_M24558_IG05 8B410C moveax, dword ptr [rcx+12] 3B420C cmpeax, dword ptr [rdx+12] 0F94C0 sete al 0FB6C0 movzxrax,al ;; bbWeight=0.50 PerfScore 7.63G_M24558_IG04: C3 ret ;; bbWeight=0.50 PerfScore 0.50G_M24558_IG05: 33C0 xoreax,eax ;; bbWeight=0.50 PerfScore 0.13G_M24558_IG06: C3 ret
                arm64 assembly
                G_M24558_IG01: A9BD7BFD stp fp, lr,[sp,#-48]! 910003FD mov fp,sp F90013A0 str x0,[fp,#32] F90017A1 str x1,[fp,#40] F9000BA2 str x2,[fp,#16] F9000FA3 str x3,[fp,#24] ;; bbWeight=1 PerfScore 5.50G_M24558_IG02: B94023A0 ldr w0,[fp,#32] B94013A1 ldr w1,[fp,#16] 6B01001F cmp w0, w1 540002A1 bne G_M24558_IG05 ;; bbWeight=1 PerfScore 5.50G_M24558_IG03: 910083A0 add x0, fp, #32 B9400400 ldr w0,[x0,#4] 910043A1 add x1, fp, #16 B9400421 ldr w1,[x1,#4] 6B01001F cmp w0, w1 540001E1 bne G_M24558_IG05 910083A0 add x0, fp, #32 B9400800 ldr w0,[x0,#8] 910043A1 add x1, fp, #16 B9400821 ldr w1,[x1,#8] 6B01001F cmp w0, w154000121 bne G_M24558_IG05 910083A0 add x0, fp, #32 B9400C00 ldr w0,[x0,#12] 910043A1 add x1, fp, #16 B9400C21 ldr w1,[x1,#12] 6B01001F cmp w0, w1 9A9F17E0 cset x0, eq ;; bbWeight=0.50 PerfScore 12.50G_M24558_IG04: A8C37BFD ldp fp, lr,[sp],#48 D65F03C0 ret lr ;; bbWeight=0.50 PerfScore 1.00G_M24558_IG05:52800000mov w0, #0 ;; bbWeight=0.50 PerfScore 0.25G_M24558_IG06: A8C37BFD ldp fp, lr,[sp],#48 D65F03C0 ret lr
                Possible arm64 assembly after fixing address calculations
                G_M24558_IG01: stp fp, lr,[sp,#-48]!mov fp,spstr x0,[fp,#32]str x1,[fp,#40]str x2,[fp,#16]str x3,[fp,#24]G_M24558_IG02: ldr w0,[fp,#32] ldr w1,[fp,#16]cmp w0, w1 bne G_M24558_IG05G_M24558_IG03: ldr w0,[fp,#36] ldr w1,[fp,#20]cmp w0, w1 bne G_M24558_IG05 ldr w0,[fp,#40] ldr w1,[fp,#24]cmp w0, w1 bne G_M24558_IG05 ldr w0,[fp,#44] ldr w1,[fp,#28]cmp w0, w1 cset x0, eqG_M24558_IG04: ldp fp, lr,[sp],#48ret lrG_M24558_IG05:mov w0, #0G_M24558_IG06: ldp fp, lr,[sp],#48ret lr

                The JIT shouldn't need to put the argument structs on the stack at all. In which case we could generate code like the following (also assuming we can compare full registers, and not 4 bytes at a time).

                Possible arm64 assembly fully optimized
                G_M24558_IG01: stp fp, lr,[sp,#-16]!mov fp,spG_M24558_IG02:cmp x0, x2 bne G_M24558_IG05cmp x1, x3 cset x0, eqG_M24558_IG04: ldp fp, lr,[sp],#16ret lrG_M24558_IG05:mov w0, #0G_M24558_IG06: ldp fp, lr,[sp],#16ret lr

                category:cq
                theme:optimization
                skill-level:intermediate
                cost:medium

                Metadata

                Metadata

                Assignees

                Labels

                Type

                No type

                Projects

                No projects

                  Relationships

                  None yet

                  Development

                  No branches or pull requests

                  Issue actions