') + ')', 'gi'); if (regex.test(text)) { found = true; var frag = document.createDocumentFragment(); var parts = text.split(regex); parts.forEach(function(part, i) { if (i % 2 === 0) { frag.appendChild(document.createTextNode(part)); } else { var span = document.createElement('span'); span.className = 'userscript-highlight'; span.textContent = part; frag.appendChild(span); } }); node.parentNode.replaceChild(frag, node); } }); } else if (node.nodeType === 1 && node.childNodes) { // element var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT']; if (!skipTags.includes(node.tagName)) { Array.from(node.childNodes).forEach(highlight); } } } highlight(document.body); // Re-highlight on dynamic content var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1 || node.nodeType === 3) highlight(node); }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ', 'i'); if (__m === '*' || __re.test(location.href)) { // Strip utm_, fbclid, gclid, etc. from all links on page (function() { var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content', 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid', 'ref', 'ref_src', 'source', 'medium', 'campaign']; function cleanUrl(url) { try { var u = new URL(url, window.location.origin); var changed = false; trackingParams.forEach(function(p) { if (u.searchParams.has(p)) { u.searchParams.delete(p); changed = true; } }); return changed ? u.toString() : url; } catch (e) { return url; } } function cleanLinks() { document.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } cleanLinks(); var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1) { if (node.tagName === 'A') cleanLinks(); node.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + ', 'i'); if (__m === '*' || __re.test(location.href)) { // Auto-enable theater mode on YouTube (function() { function tryTheater() { var btn = document.querySelector('button[aria-label="Theater mode"], ytd-player #player button[title="Theater mode"]'); if (btn && !btn.classList.contains('activated')) { btn.click(); } } // Try immediately tryTheater(); // Try after navigation (SPA) var lastUrl = location.href; setInterval(function() { if (location.href !== lastUrl) { lastUrl = location.href; setTimeout(tryTheater, 500); } }, 1000); // Also try on player load var observer = new MutationObserver(tryTheater); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ', 'i'); if (__m === '*' || __re.test(location.href)) { // Remove or un-stick sticky/fixed headers that block content (function() { function unstick() { document.querySelectorAll('header, nav, [role="banner"], .header, .navbar, .sticky, .fixed-top, [style*="position: fixed"], [style*="position:sticky"]').forEach(function(el) { if (el.style.position === 'fixed' || el.style.position === 'sticky' || getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') { el.style.position = 'static'; el.style.top = 'auto'; el.style.zIndex = 'auto'; } }); } unstick(); var observer = new MutationObserver(unstick); observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] }); })(); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); })(); test(cu): harden real provider evidence by hqhq1025 · Pull Request #926 · apache/maka · GitHub
Skip to content

test(cu): harden real provider evidence - #926

Merged
Astro-Han merged 1 commit into
apache:mainfrom
hqhq1025:codex/cu-evidence-hardening
Jul 14, 2026
Merged

test(cu): harden real provider evidence#926
Astro-Han merged 1 commit into
apache:mainfrom
hqhq1025:codex/cu-evidence-hardening

Conversation

@hqhq1025

@hqhq1025hqhq1025 commented Jul 13, 2026

Copy link
Copy Markdown
Contributor

Summary

Addresses every evidence-layer P2/P3 from the #913 review on top of merged #924.

  • one schema-driven sanitizer for direct and Desktop reports
  • validated producer/provider/model/transport attribution; no raw outer errors, prompts, screenshots, UI content, or arbitrary trace strings
  • canonical scenario assertions in the provider matrix
  • real reports require schema v1, real-runtime, status pass, matching provider/model/scenario, complete/end_turn, successful owned actions, budgets, fixture/forbidden evidence, and AX/semantic dispatch proof
  • missing forbidden evidence is inconclusive; malformed violations cannot crash the matrix
  • real-model policy is mandatory and enforces exact fixture selectors, owned observation lineage, per-action budgets, and total budget before dispatch
  • ordinary launcher fails closed for dedicated runners and unavailable capabilities
  • stale-window fixture enumerates surviving windows; partial fixture construction tears down prior windows
  • contradictory scenario budgets are rejected

Real validation

OpenAI gpt-5.4 passed the hardened qualification gate:

  • L0: one successful owned observe, exact fixture oracle, complete/end_turn
  • L1: two successful owned observes + one owned AX click_element, primary=1, danger=0, duplicate=0, AX dispatch evidence

Verification

@hqhq1025

Copy link
Copy Markdown
ContributorAuthor

@Astro-Han Addressed all 14 review findings from merged #913 on top of #924. The report sanitizer, matrix validator, runtime policy, fixture lifecycle, runner/capability gates, budgets, owned-target proof, terminal qualification, and AX/semantic dispatch proof are now fail-closed. CI is fully green; hardened OpenAI gpt-5.4 L0 and L1 reruns pass. Ready for review.

@Astro-HanAstro-Han left a comment

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

No product-level P0/P1 findings. The issues below affect the reliability of the real-provider evidence harness, so I am treating them as non-blocking P2s:

  • P2: The stale-window scenario builds its allowlist from the initial window set, so the replacement Current Window can be rejected.
  • P2: targetOwned can be decided by weaker conditions before fixture-PID trace binding runs.
  • P2: The L1 gate accepts fewer observations than the scenario prompt requires and evaluates only the qualifying subset of actions.
  • P2: Direct AppKit AX reports do not satisfy the new canonical report schema.
  • P2: wait and cursor_position are treated as observation-bound actions even though their schemas do not accept an observation ID.
  • P2: Driver traces are appended asynchronously without a flush before qualification reads them.
  • P2: Producer and policy provenance validation remains fail-open when fields are absent or unexpected.

@Astro-Han
Astro-Han merged commit 08163f1 into apache:mainJul 14, 2026
3 checks passed
Sign up for freeto join this conversation on GitHub. Already have an account? Sign in to comment

Labels

None yet

Projects

None yet

Development

Successfully merging this pull request may close these issues.

2 participants

@hqhq1025@Astro-Han