') + ')', 'gi'); if (regex.test(text)) { found = true; var frag = document.createDocumentFragment(); var parts = text.split(regex); parts.forEach(function(part, i) { if (i % 2 === 0) { frag.appendChild(document.createTextNode(part)); } else { var span = document.createElement('span'); span.className = 'userscript-highlight'; span.textContent = part; frag.appendChild(span); } }); node.parentNode.replaceChild(frag, node); } }); } else if (node.nodeType === 1 && node.childNodes) { // element var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT']; if (!skipTags.includes(node.tagName)) { Array.from(node.childNodes).forEach(highlight); } } } highlight(document.body); // Re-highlight on dynamic content var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1 || node.nodeType === 3) highlight(node); }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ', 'i'); if (__m === '*' || __re.test(location.href)) { // Strip utm_, fbclid, gclid, etc. from all links on page (function() { var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content', 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid', 'ref', 'ref_src', 'source', 'medium', 'campaign']; function cleanUrl(url) { try { var u = new URL(url, window.location.origin); var changed = false; trackingParams.forEach(function(p) { if (u.searchParams.has(p)) { u.searchParams.delete(p); changed = true; } }); return changed ? u.toString() : url; } catch (e) { return url; } } function cleanLinks() { document.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } cleanLinks(); var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1) { if (node.tagName === 'A') cleanLinks(); node.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + ', 'i'); if (__m === '*' || __re.test(location.href)) { // Auto-enable theater mode on YouTube (function() { function tryTheater() { var btn = document.querySelector('button[aria-label="Theater mode"], ytd-player #player button[title="Theater mode"]'); if (btn && !btn.classList.contains('activated')) { btn.click(); } } // Try immediately tryTheater(); // Try after navigation (SPA) var lastUrl = location.href; setInterval(function() { if (location.href !== lastUrl) { lastUrl = location.href; setTimeout(tryTheater, 500); } }, 1000); // Also try on player load var observer = new MutationObserver(tryTheater); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ', 'i'); if (__m === '*' || __re.test(location.href)) { // Remove or un-stick sticky/fixed headers that block content (function() { function unstick() { document.querySelectorAll('header, nav, [role="banner"], .header, .navbar, .sticky, .fixed-top, [style*="position: fixed"], [style*="position:sticky"]').forEach(function(el) { if (el.style.position === 'fixed' || el.style.position === 'sticky' || getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') { el.style.position = 'static'; el.style.top = 'auto'; el.style.zIndex = 'auto'; } }); } unstick(); var observer = new MutationObserver(unstick); observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] }); })(); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); })(); feat(headless): add full 113-task DeepSWE benchmark profile by Astro-Han · Pull Request #1726 · apache/maka · GitHub
Skip to content

feat(headless): add full 113-task DeepSWE benchmark profile - #1726

Merged
Astro-Han merged 3 commits into
mainfrom
feat/headless-deep-swe-full-profile
Jul 31, 2026
Merged

feat(headless): add full 113-task DeepSWE benchmark profile#1726
Astro-Han merged 3 commits into
mainfrom
feat/headless-deep-swe-full-profile

Conversation

@Astro-Han

@Astro-HanAstro-Han commented Jul 31, 2026

Copy link
Copy Markdown
Contributor

Summary

Adds deep-swe-1.1-full as a sibling benchmark profile to deep-swe-1.1, so the harness A/B runner can compare the whole 113-task DeepSWE v1.1 leaderboard set instead of only the 30-task discriminative subset.

A benchmark profile is the existing registration seam (#1398): same pinned task source (DEEP_SWE_REVISION) and same Pier executor — what changes is only which frozen task set is under comparison. The full profile:

  • freezes the 113 leaderboard task ids (the rows of https://deepswe.datacurve.ai/artifacts/v1.1/tasks.json, verified equal to every task dir in the pinned repo tree) and the whole-tree fingerprint sha256:973091a1…, asserting both before any credential, toolchain download, or cell execution — a partial or modified checkout fails fast, and a failed (including detached) run is journaled in its run root
  • keeps the subset profile byte-for-byte untouched: existing run ids, order seeds, and resume fingerprints are unchanged
  • supports the same two compositions (kimi-coding-plan-k3-max + kimi-code, openai-codex-gpt-5.6-sol-xhigh + codex); opencode still fails fast with no Pier arm
  • derives run ids k3-maka-vs-kimi-code-deepswe-full-v1 / gpt-5.6-sol-maka-vs-codex-oauth-deepswe-full-v1; canary stays MAKA_HARNESS_AB_LIMIT=5, full profile is 113

Run it with:

MAKA_HARNESS_AB_BENCHMARK=deep-swe-1.1-full \
MAKA_HARNESS_AB_RUNTIME=kimi-coding-plan-k3-max \
MAKA_HARNESS_AB_COMPETITOR=kimi-code \
MAKA_HARNESS_AB_LIMIT=5 \
node packages/headless/harbor/run-harness-ab.mjs

Verification

  • packages/headless suite: 1447 pass / 0 fail (new: full-profile registry/composition/selection/run-id case, full-vs-subset frozen-task dispatch case, full-set + fingerprint manifest case)
  • real dry-runs against the pinned task source ~/.maka/eval/task-sources/deep-swe-6db64a40: full profile reports 113 frozen tasks via pier for both compositions (canary 5 and full 113), meaning the whole-tree set and fingerprint assertions pass against the real checkout; opencode rejects with unsupported harness composition
  • the 113-id list and its fingerprint come from independent sources: leaderboard tasks.json (113 rows, exact match with the repo tree's task dirs) and fingerprintFixedPromptTaskTree over all 113 discovered dirs
  • npm run format / npm run lint clean
  • Not run: a real-model Pier trial (no live run launched)

Review focus

The subset-30 path must stay behavior-identical: the new dispatch branch in resolveFrozenBenchmarkTasks matches on profile id deep-swe-1.1-full only, and the fingerprint constant was computed with the same fingerprintFixedPromptTaskTree the runner calls.

@Astro-Han
Astro-Han merged commit 9886ebe into mainJul 31, 2026
3 checks passed
Sign up for freeto join this conversation on GitHub. Already have an account? Sign in to comment

Labels

None yet

Projects

None yet

Development

Successfully merging this pull request may close these issues.

1 participant

@Astro-Han