Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 28 additions & 0 deletions docs/decision-semantics.md
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,28 @@
# Evaluation Decision Semantics

This framework keeps three concepts separate:

| Concept | Meaning | Where it is recorded |
|---|---|---|
| System risk tier | Overall impact context for the evaluated system and use case | `metadata.risk_tier` (`low`, `medium`, `high`) |
| Scenario or finding severity | Consequence if one specific scenario or failure occurs | `scenarios[].risk` and `findings.failures[].severity` (`low`, `medium`, `high`, `critical`) |
| Decision blocker | An unresolved condition that prevents the current release decision | `recommendation.blockers` |

A high- or critical-severity scenario does not automatically define the whole system's risk tier. Conversely, a high-tier system may include routine lower-severity test scenarios.

## Recommendation fields

- Use `recommendation.blockers` only for unresolved items that prevent the current decision.
- Use `recommendation.required_actions` for follow-up work accepted as part of a bounded conditional approval.
- Use `decision.conditions` for the named, accountable constraints of that conditional approval.

A report with `release_with_conditions` must not claim unresolved blockers. It must instead state the required actions and release conditions that named owners have accepted.

## Validator rules

- A plain `release` cannot include blockers, required actions, or conditions.
- A high-severity finding prevents an unconditional `release`.
- A critical finding requires `do_not_release`.
- `do_not_release` requires either a stated blocker or at least one critical finding.

These rules make report logic internally coherent. They are practitioner controls, not a substitute for organization-specific release authority, legal review, safety analysis, or compliance decisions.
6 changes: 3 additions & 3 deletions examples/sample-evaluation-report.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -10,8 +10,8 @@
},
"recommendation": {
"decision": "release_with_conditions",
"reason": "The agent met core task-completion and routing-quality targets, but failed one high-risk escalation scenario involving sensitive complaint handling.",
"blockers": [
"reason": "The agent met core task-completion and routing-quality targets, but failed one high-severity escalation scenario involving sensitive complaint handling.",
"required_actions": [
"Add deterministic escalation rule for personal-data and legal-threat complaints",
"Re-run high-risk escalation scenario pack before production release"
]
Expand DownExpand Up@@ -167,4 +167,4 @@
}
]
}
}
}
11 changes: 8 additions & 3 deletions schemas/evaluation-report.schema.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -54,6 +54,11 @@
"type": "array",
"items": { "type": "string", "minLength": 5, "maxLength": 300 },
"uniqueItems": true
},
"required_actions": {
"type": "array",
"items": { "type": "string", "minLength": 5, "maxLength": 300 },
"uniqueItems": true
}
}
},
Expand DownExpand Up@@ -83,7 +88,7 @@
"properties": {
"id": { "type": "string", "pattern": "^S-\\d{3}$" },
"scenario": { "type": "string", "minLength": 5, "maxLength": 250 },
"risk": { "type": "string", "enum": ["low", "medium", "high"] },
"risk": { "type": "string", "enum": ["low", "medium", "high", "critical"] },
"expected_behavior": { "type": "string", "minLength": 5, "maxLength": 400 },
"result": { "type": "string", "enum": ["pass", "fail", "partial"] },
"notes": { "type": "string", "maxLength": 500 }
Expand DownExpand Up@@ -123,7 +128,7 @@
"required": ["finding", "severity", "evidence", "required_action"],
"properties": {
"finding": { "type": "string", "minLength": 5, "maxLength": 300 },
"severity": { "type": "string", "enum": ["low", "medium", "high"] },
"severity": { "type": "string", "enum": ["low", "medium", "high", "critical"] },
"evidence": { "type": "string", "minLength": 2, "maxLength": 120 },
"required_action": { "type": "string", "minLength": 5, "maxLength": 300 }
}
Expand DownExpand Up@@ -175,4 +180,4 @@
}
}
}
}
}
38 changes: 33 additions & 5 deletions tests/test_validate_evaluation_report.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -27,35 +27,63 @@ def test_sample_report_is_semantically_coherent(self) -> None:
def test_release_cannot_claim_not_approved_status(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["decision"] = "release"
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["release_status"] = "not_approved"
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("requires decision.release_status=approved" in error for error in errors))

def test_release_with_conditions_requires_conditions_and_blockers(self) -> None:
def test_conditional_release_requires_conditions_or_required_actions(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("requires at least one stated blocker" in error for error in errors))
self.assertTrue(any("requires at least one stated required_action" in error for error in errors))
self.assertTrue(any("requires at least one release condition" in error for error in errors))

def test_conditional_release_cannot_claim_unresolved_blockers(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["blockers"] = ["Resolve the fictional security boundary"]

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("incompatible with unresolved blockers" in error for error in errors))

def test_release_cannot_ignore_high_severity_finding(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["decision"] = "release"
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["release_status"] = "approved"
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("high-severity findings" in error for error in errors))

def test_critical_finding_requires_do_not_release(self) -> None:
report = copy.deepcopy(self.report)
report["findings"]["failures"][0]["severity"] = "critical"

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("critical findings require" in error for error in errors))

def test_do_not_release_can_be_supported_by_critical_finding(self) -> None:
report = copy.deepcopy(self.report)
report["findings"]["failures"][0]["severity"] = "critical"
report["recommendation"] = {
"decision": "do_not_release",
"reason": "A fictional critical boundary failure remains unresolved.",
}
report["decision"]["release_status"] = "not_approved"
report["decision"]["conditions"] = []

self.assertEqual(VALIDATOR.semantic_errors(report), [])


if __name__ == "__main__":
unittest.main()
40 changes: 28 additions & 12 deletions tools/validate_evaluation_report.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,7 +3,7 @@

The JSON Schema checks field types and allowed values. This utility adds the
cross-field rules needed to keep a recommendation, release status, conditions,
and high-severity findings internally coherent.
required actions, decision blockers, and material findings internally coherent.
"""

from __future__ import annotations
Expand DownExpand Up@@ -50,7 +50,13 @@ def schema_errors(schema: dict[str, Any], report: dict[str, Any]) -> list[str]:


def semantic_errors(report: dict[str, Any]) -> list[str]:
"""Return cross-field consistency errors after schema validation succeeds."""
"""Return cross-field consistency errors after schema validation succeeds.

A *severity* states the consequence of a finding. A *blocker* is a decision
status: unresolved blockers prevent the current release decision. Required
actions may exist on a conditional approval, but they are not blockers once
the accountable owners have accepted the bounded conditions.
"""
recommendation = report["recommendation"]
decision = report["decision"]
recommendation_decision = recommendation["decision"]
Expand All@@ -67,30 +73,40 @@ def semantic_errors(report: dict[str, Any]) -> list[str]:
)

blockers = recommendation.get("blockers", [])
required_actions = recommendation.get("required_actions", [])
conditions = decision["conditions"]
severities = [failure["severity"] for failure in report["findings"]["failures"]]
has_high = "high" in severities
has_critical = "critical" in severities

if recommendation_decision == "release":
if blockers:
errors.append("recommendation.decision=release must not include unresolved blockers")
if required_actions:
errors.append("recommendation.decision=release must not include unresolved required_actions")
if conditions:
errors.append("decision.release_status=approved must not include release conditions")

if recommendation_decision == "release_with_conditions":
if not blockers:
errors.append("release_with_conditions requires at least one stated blocker or required action")
if blockers:
errors.append(
"release_with_conditions is incompatible with unresolved blockers; "
"use required_actions and decision.conditions for bounded follow-up"
)
if not required_actions and not conditions:
errors.append(
"release_with_conditions requires at least one stated required_action or release condition"
)
if not conditions:
errors.append("approved_with_conditions requires at least one release condition")

if recommendation_decision == "do_not_release" and not blockers:
errors.append("do_not_release requires at least one stated blocker")
if recommendation_decision == "do_not_release" and not blockers and not has_critical:
errors.append("do_not_release requires at least one stated blocker or critical finding")

high_findings = [
failure
for failure in report["findings"]["failures"]
if failure["severity"] == "high"
]
if high_findings and recommendation_decision == "release":
if has_high and recommendation_decision == "release":
errors.append("recommendation.decision=release is incompatible with unresolved high-severity findings")
if has_critical and recommendation_decision != "do_not_release":
errors.append("critical findings require recommendation.decision=do_not_release")

return errors

Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 28 additions & 0 deletions docs/decision-semantics.md
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,28 @@
# Evaluation Decision Semantics

This framework keeps three concepts separate:

| Concept | Meaning | Where it is recorded |
|---|---|---|
| System risk tier | Overall impact context for the evaluated system and use case | `metadata.risk_tier` (`low`, `medium`, `high`) |
| Scenario or finding severity | Consequence if one specific scenario or failure occurs | `scenarios[].risk` and `findings.failures[].severity` (`low`, `medium`, `high`, `critical`) |
| Decision blocker | An unresolved condition that prevents the current release decision | `recommendation.blockers` |

A high- or critical-severity scenario does not automatically define the whole system's risk tier. Conversely, a high-tier system may include routine lower-severity test scenarios.

## Recommendation fields

- Use `recommendation.blockers` only for unresolved items that prevent the current decision.
- Use `recommendation.required_actions` for follow-up work accepted as part of a bounded conditional approval.
- Use `decision.conditions` for the named, accountable constraints of that conditional approval.

A report with `release_with_conditions` must not claim unresolved blockers. It must instead state the required actions and release conditions that named owners have accepted.

## Validator rules

- A plain `release` cannot include blockers, required actions, or conditions.
- A high-severity finding prevents an unconditional `release`.
- A critical finding requires `do_not_release`.
- `do_not_release` requires either a stated blocker or at least one critical finding.

These rules make report logic internally coherent. They are practitioner controls, not a substitute for organization-specific release authority, legal review, safety analysis, or compliance decisions.
6 changes: 3 additions & 3 deletions examples/sample-evaluation-report.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -10,8 +10,8 @@
},
"recommendation": {
"decision": "release_with_conditions",
"reason": "The agent met core task-completion and routing-quality targets, but failed one high-risk escalation scenario involving sensitive complaint handling.",
"blockers": [
"reason": "The agent met core task-completion and routing-quality targets, but failed one high-severity escalation scenario involving sensitive complaint handling.",
"required_actions": [
"Add deterministic escalation rule for personal-data and legal-threat complaints",
"Re-run high-risk escalation scenario pack before production release"
]
Expand DownExpand Up@@ -167,4 +167,4 @@
}
]
}
}
}
11 changes: 8 additions & 3 deletions schemas/evaluation-report.schema.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -54,6 +54,11 @@
"type": "array",
"items": { "type": "string", "minLength": 5, "maxLength": 300 },
"uniqueItems": true
},
"required_actions": {
"type": "array",
"items": { "type": "string", "minLength": 5, "maxLength": 300 },
"uniqueItems": true
}
}
},
Expand DownExpand Up@@ -83,7 +88,7 @@
"properties": {
"id": { "type": "string", "pattern": "^S-\\d{3}$" },
"scenario": { "type": "string", "minLength": 5, "maxLength": 250 },
"risk": { "type": "string", "enum": ["low", "medium", "high"] },
"risk": { "type": "string", "enum": ["low", "medium", "high", "critical"] },
"expected_behavior": { "type": "string", "minLength": 5, "maxLength": 400 },
"result": { "type": "string", "enum": ["pass", "fail", "partial"] },
"notes": { "type": "string", "maxLength": 500 }
Expand DownExpand Up@@ -123,7 +128,7 @@
"required": ["finding", "severity", "evidence", "required_action"],
"properties": {
"finding": { "type": "string", "minLength": 5, "maxLength": 300 },
"severity": { "type": "string", "enum": ["low", "medium", "high"] },
"severity": { "type": "string", "enum": ["low", "medium", "high", "critical"] },
"evidence": { "type": "string", "minLength": 2, "maxLength": 120 },
"required_action": { "type": "string", "minLength": 5, "maxLength": 300 }
}
Expand DownExpand Up@@ -175,4 +180,4 @@
}
}
}
}
}
38 changes: 33 additions & 5 deletions tests/test_validate_evaluation_report.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -27,35 +27,63 @@ def test_sample_report_is_semantically_coherent(self) -> None:
def test_release_cannot_claim_not_approved_status(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["decision"] = "release"
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["release_status"] = "not_approved"
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("requires decision.release_status=approved" in error for error in errors))

def test_release_with_conditions_requires_conditions_and_blockers(self) -> None:
def test_conditional_release_requires_conditions_or_required_actions(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("requires at least one stated blocker" in error for error in errors))
self.assertTrue(any("requires at least one stated required_action" in error for error in errors))
self.assertTrue(any("requires at least one release condition" in error for error in errors))

def test_conditional_release_cannot_claim_unresolved_blockers(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["blockers"] = ["Resolve the fictional security boundary"]

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("incompatible with unresolved blockers" in error for error in errors))

def test_release_cannot_ignore_high_severity_finding(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["decision"] = "release"
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["release_status"] = "approved"
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("high-severity findings" in error for error in errors))

def test_critical_finding_requires_do_not_release(self) -> None:
report = copy.deepcopy(self.report)
report["findings"]["failures"][0]["severity"] = "critical"

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("critical findings require" in error for error in errors))

def test_do_not_release_can_be_supported_by_critical_finding(self) -> None:
report = copy.deepcopy(self.report)
report["findings"]["failures"][0]["severity"] = "critical"
report["recommendation"] = {
"decision": "do_not_release",
"reason": "A fictional critical boundary failure remains unresolved.",
}
report["decision"]["release_status"] = "not_approved"
report["decision"]["conditions"] = []

self.assertEqual(VALIDATOR.semantic_errors(report), [])


if __name__ == "__main__":
unittest.main()
40 changes: 28 additions & 12 deletions tools/validate_evaluation_report.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,7 +3,7 @@

The JSON Schema checks field types and allowed values. This utility adds the
cross-field rules needed to keep a recommendation, release status, conditions,
and high-severity findings internally coherent.
required actions, decision blockers, and material findings internally coherent.
"""

from __future__ import annotations
Expand DownExpand Up@@ -50,7 +50,13 @@ def schema_errors(schema: dict[str, Any], report: dict[str, Any]) -> list[str]:


def semantic_errors(report: dict[str, Any]) -> list[str]:
"""Return cross-field consistency errors after schema validation succeeds."""
"""Return cross-field consistency errors after schema validation succeeds.

A *severity* states the consequence of a finding. A *blocker* is a decision
status: unresolved blockers prevent the current release decision. Required
actions may exist on a conditional approval, but they are not blockers once
the accountable owners have accepted the bounded conditions.
"""
recommendation = report["recommendation"]
decision = report["decision"]
recommendation_decision = recommendation["decision"]
Expand All@@ -67,30 +73,40 @@ def semantic_errors(report: dict[str, Any]) -> list[str]:
)

blockers = recommendation.get("blockers", [])
required_actions = recommendation.get("required_actions", [])
conditions = decision["conditions"]
severities = [failure["severity"] for failure in report["findings"]["failures"]]
has_high = "high" in severities
has_critical = "critical" in severities

if recommendation_decision == "release":
if blockers:
errors.append("recommendation.decision=release must not include unresolved blockers")
if required_actions:
errors.append("recommendation.decision=release must not include unresolved required_actions")
if conditions:
errors.append("decision.release_status=approved must not include release conditions")

if recommendation_decision == "release_with_conditions":
if not blockers:
errors.append("release_with_conditions requires at least one stated blocker or required action")
if blockers:
errors.append(
"release_with_conditions is incompatible with unresolved blockers; "
"use required_actions and decision.conditions for bounded follow-up"
)
if not required_actions and not conditions:
errors.append(
"release_with_conditions requires at least one stated required_action or release condition"
)
if not conditions:
errors.append("approved_with_conditions requires at least one release condition")

if recommendation_decision == "do_not_release" and not blockers:
errors.append("do_not_release requires at least one stated blocker")
if recommendation_decision == "do_not_release" and not blockers and not has_critical:
errors.append("do_not_release requires at least one stated blocker or critical finding")

high_findings = [
failure
for failure in report["findings"]["failures"]
if failure["severity"] == "high"
]
if high_findings and recommendation_decision == "release":
if has_high and recommendation_decision == "release":
errors.append("recommendation.decision=release is incompatible with unresolved high-severity findings")
if has_critical and recommendation_decision != "do_not_release":
errors.append("critical findings require recommendation.decision=do_not_release")

return errors

Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 28 additions & 0 deletions docs/decision-semantics.md
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,28 @@
# Evaluation Decision Semantics

This framework keeps three concepts separate:

| Concept | Meaning | Where it is recorded |
|---|---|---|
| System risk tier | Overall impact context for the evaluated system and use case | `metadata.risk_tier` (`low`, `medium`, `high`) |
| Scenario or finding severity | Consequence if one specific scenario or failure occurs | `scenarios[].risk` and `findings.failures[].severity` (`low`, `medium`, `high`, `critical`) |
| Decision blocker | An unresolved condition that prevents the current release decision | `recommendation.blockers` |

A high- or critical-severity scenario does not automatically define the whole system's risk tier. Conversely, a high-tier system may include routine lower-severity test scenarios.

## Recommendation fields

- Use `recommendation.blockers` only for unresolved items that prevent the current decision.
- Use `recommendation.required_actions` for follow-up work accepted as part of a bounded conditional approval.
- Use `decision.conditions` for the named, accountable constraints of that conditional approval.

A report with `release_with_conditions` must not claim unresolved blockers. It must instead state the required actions and release conditions that named owners have accepted.

## Validator rules

- A plain `release` cannot include blockers, required actions, or conditions.
- A high-severity finding prevents an unconditional `release`.
- A critical finding requires `do_not_release`.
- `do_not_release` requires either a stated blocker or at least one critical finding.

These rules make report logic internally coherent. They are practitioner controls, not a substitute for organization-specific release authority, legal review, safety analysis, or compliance decisions.
6 changes: 3 additions & 3 deletions examples/sample-evaluation-report.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -10,8 +10,8 @@
},
"recommendation": {
"decision": "release_with_conditions",
"reason": "The agent met core task-completion and routing-quality targets, but failed one high-risk escalation scenario involving sensitive complaint handling.",
"blockers": [
"reason": "The agent met core task-completion and routing-quality targets, but failed one high-severity escalation scenario involving sensitive complaint handling.",
"required_actions": [
"Add deterministic escalation rule for personal-data and legal-threat complaints",
"Re-run high-risk escalation scenario pack before production release"
]
Expand DownExpand Up@@ -167,4 +167,4 @@
}
]
}
}
}
11 changes: 8 additions & 3 deletions schemas/evaluation-report.schema.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -54,6 +54,11 @@
"type": "array",
"items": { "type": "string", "minLength": 5, "maxLength": 300 },
"uniqueItems": true
},
"required_actions": {
"type": "array",
"items": { "type": "string", "minLength": 5, "maxLength": 300 },
"uniqueItems": true
}
}
},
Expand DownExpand Up@@ -83,7 +88,7 @@
"properties": {
"id": { "type": "string", "pattern": "^S-\\d{3}$" },
"scenario": { "type": "string", "minLength": 5, "maxLength": 250 },
"risk": { "type": "string", "enum": ["low", "medium", "high"] },
"risk": { "type": "string", "enum": ["low", "medium", "high", "critical"] },
"expected_behavior": { "type": "string", "minLength": 5, "maxLength": 400 },
"result": { "type": "string", "enum": ["pass", "fail", "partial"] },
"notes": { "type": "string", "maxLength": 500 }
Expand DownExpand Up@@ -123,7 +128,7 @@
"required": ["finding", "severity", "evidence", "required_action"],
"properties": {
"finding": { "type": "string", "minLength": 5, "maxLength": 300 },
"severity": { "type": "string", "enum": ["low", "medium", "high"] },
"severity": { "type": "string", "enum": ["low", "medium", "high", "critical"] },
"evidence": { "type": "string", "minLength": 2, "maxLength": 120 },
"required_action": { "type": "string", "minLength": 5, "maxLength": 300 }
}
Expand DownExpand Up@@ -175,4 +180,4 @@
}
}
}
}
}
38 changes: 33 additions & 5 deletions tests/test_validate_evaluation_report.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -27,35 +27,63 @@ def test_sample_report_is_semantically_coherent(self) -> None:
def test_release_cannot_claim_not_approved_status(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["decision"] = "release"
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["release_status"] = "not_approved"
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("requires decision.release_status=approved" in error for error in errors))

def test_release_with_conditions_requires_conditions_and_blockers(self) -> None:
def test_conditional_release_requires_conditions_or_required_actions(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("requires at least one stated blocker" in error for error in errors))
self.assertTrue(any("requires at least one stated required_action" in error for error in errors))
self.assertTrue(any("requires at least one release condition" in error for error in errors))

def test_conditional_release_cannot_claim_unresolved_blockers(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["blockers"] = ["Resolve the fictional security boundary"]

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("incompatible with unresolved blockers" in error for error in errors))

def test_release_cannot_ignore_high_severity_finding(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["decision"] = "release"
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["release_status"] = "approved"
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("high-severity findings" in error for error in errors))

def test_critical_finding_requires_do_not_release(self) -> None:
report = copy.deepcopy(self.report)
report["findings"]["failures"][0]["severity"] = "critical"

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("critical findings require" in error for error in errors))

def test_do_not_release_can_be_supported_by_critical_finding(self) -> None:
report = copy.deepcopy(self.report)
report["findings"]["failures"][0]["severity"] = "critical"
report["recommendation"] = {
"decision": "do_not_release",
"reason": "A fictional critical boundary failure remains unresolved.",
}
report["decision"]["release_status"] = "not_approved"
report["decision"]["conditions"] = []

self.assertEqual(VALIDATOR.semantic_errors(report), [])


if __name__ == "__main__":
unittest.main()
40 changes: 28 additions & 12 deletions tools/validate_evaluation_report.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,7 +3,7 @@

The JSON Schema checks field types and allowed values. This utility adds the
cross-field rules needed to keep a recommendation, release status, conditions,
and high-severity findings internally coherent.
required actions, decision blockers, and material findings internally coherent.
"""

from __future__ import annotations
Expand DownExpand Up@@ -50,7 +50,13 @@ def schema_errors(schema: dict[str, Any], report: dict[str, Any]) -> list[str]:


def semantic_errors(report: dict[str, Any]) -> list[str]:
"""Return cross-field consistency errors after schema validation succeeds."""
"""Return cross-field consistency errors after schema validation succeeds.

A *severity* states the consequence of a finding. A *blocker* is a decision
status: unresolved blockers prevent the current release decision. Required
actions may exist on a conditional approval, but they are not blockers once
the accountable owners have accepted the bounded conditions.
"""
recommendation = report["recommendation"]
decision = report["decision"]
recommendation_decision = recommendation["decision"]
Expand All@@ -67,30 +73,40 @@ def semantic_errors(report: dict[str, Any]) -> list[str]:
)

blockers = recommendation.get("blockers", [])
required_actions = recommendation.get("required_actions", [])
conditions = decision["conditions"]
severities = [failure["severity"] for failure in report["findings"]["failures"]]
has_high = "high" in severities
has_critical = "critical" in severities

if recommendation_decision == "release":
if blockers:
errors.append("recommendation.decision=release must not include unresolved blockers")
if required_actions:
errors.append("recommendation.decision=release must not include unresolved required_actions")
if conditions:
errors.append("decision.release_status=approved must not include release conditions")

if recommendation_decision == "release_with_conditions":
if not blockers:
errors.append("release_with_conditions requires at least one stated blocker or required action")
if blockers:
errors.append(
"release_with_conditions is incompatible with unresolved blockers; "
"use required_actions and decision.conditions for bounded follow-up"
)
if not required_actions and not conditions:
errors.append(
"release_with_conditions requires at least one stated required_action or release condition"
)
if not conditions:
errors.append("approved_with_conditions requires at least one release condition")

if recommendation_decision == "do_not_release" and not blockers:
errors.append("do_not_release requires at least one stated blocker")
if recommendation_decision == "do_not_release" and not blockers and not has_critical:
errors.append("do_not_release requires at least one stated blocker or critical finding")

high_findings = [
failure
for failure in report["findings"]["failures"]
if failure["severity"] == "high"
]
if high_findings and recommendation_decision == "release":
if has_high and recommendation_decision == "release":
errors.append("recommendation.decision=release is incompatible with unresolved high-severity findings")
if has_critical and recommendation_decision != "do_not_release":
errors.append("critical findings require recommendation.decision=do_not_release")

return errors

Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 28 additions & 0 deletions docs/decision-semantics.md
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,28 @@
# Evaluation Decision Semantics

This framework keeps three concepts separate:

| Concept | Meaning | Where it is recorded |
|---|---|---|
| System risk tier | Overall impact context for the evaluated system and use case | `metadata.risk_tier` (`low`, `medium`, `high`) |
| Scenario or finding severity | Consequence if one specific scenario or failure occurs | `scenarios[].risk` and `findings.failures[].severity` (`low`, `medium`, `high`, `critical`) |
| Decision blocker | An unresolved condition that prevents the current release decision | `recommendation.blockers` |

A high- or critical-severity scenario does not automatically define the whole system's risk tier. Conversely, a high-tier system may include routine lower-severity test scenarios.

## Recommendation fields

- Use `recommendation.blockers` only for unresolved items that prevent the current decision.
- Use `recommendation.required_actions` for follow-up work accepted as part of a bounded conditional approval.
- Use `decision.conditions` for the named, accountable constraints of that conditional approval.

A report with `release_with_conditions` must not claim unresolved blockers. It must instead state the required actions and release conditions that named owners have accepted.

## Validator rules

- A plain `release` cannot include blockers, required actions, or conditions.
- A high-severity finding prevents an unconditional `release`.
- A critical finding requires `do_not_release`.
- `do_not_release` requires either a stated blocker or at least one critical finding.

These rules make report logic internally coherent. They are practitioner controls, not a substitute for organization-specific release authority, legal review, safety analysis, or compliance decisions.
6 changes: 3 additions & 3 deletions examples/sample-evaluation-report.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -10,8 +10,8 @@
},
"recommendation": {
"decision": "release_with_conditions",
"reason": "The agent met core task-completion and routing-quality targets, but failed one high-risk escalation scenario involving sensitive complaint handling.",
"blockers": [
"reason": "The agent met core task-completion and routing-quality targets, but failed one high-severity escalation scenario involving sensitive complaint handling.",
"required_actions": [
"Add deterministic escalation rule for personal-data and legal-threat complaints",
"Re-run high-risk escalation scenario pack before production release"
]
Expand DownExpand Up@@ -167,4 +167,4 @@
}
]
}
}
}
11 changes: 8 additions & 3 deletions schemas/evaluation-report.schema.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -54,6 +54,11 @@
"type": "array",
"items": { "type": "string", "minLength": 5, "maxLength": 300 },
"uniqueItems": true
},
"required_actions": {
"type": "array",
"items": { "type": "string", "minLength": 5, "maxLength": 300 },
"uniqueItems": true
}
}
},
Expand DownExpand Up@@ -83,7 +88,7 @@
"properties": {
"id": { "type": "string", "pattern": "^S-\\d{3}$" },
"scenario": { "type": "string", "minLength": 5, "maxLength": 250 },
"risk": { "type": "string", "enum": ["low", "medium", "high"] },
"risk": { "type": "string", "enum": ["low", "medium", "high", "critical"] },
"expected_behavior": { "type": "string", "minLength": 5, "maxLength": 400 },
"result": { "type": "string", "enum": ["pass", "fail", "partial"] },
"notes": { "type": "string", "maxLength": 500 }
Expand DownExpand Up@@ -123,7 +128,7 @@
"required": ["finding", "severity", "evidence", "required_action"],
"properties": {
"finding": { "type": "string", "minLength": 5, "maxLength": 300 },
"severity": { "type": "string", "enum": ["low", "medium", "high"] },
"severity": { "type": "string", "enum": ["low", "medium", "high", "critical"] },
"evidence": { "type": "string", "minLength": 2, "maxLength": 120 },
"required_action": { "type": "string", "minLength": 5, "maxLength": 300 }
}
Expand DownExpand Up@@ -175,4 +180,4 @@
}
}
}
}
}
38 changes: 33 additions & 5 deletions tests/test_validate_evaluation_report.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -27,35 +27,63 @@ def test_sample_report_is_semantically_coherent(self) -> None:
def test_release_cannot_claim_not_approved_status(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["decision"] = "release"
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["release_status"] = "not_approved"
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("requires decision.release_status=approved" in error for error in errors))

def test_release_with_conditions_requires_conditions_and_blockers(self) -> None:
def test_conditional_release_requires_conditions_or_required_actions(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("requires at least one stated blocker" in error for error in errors))
self.assertTrue(any("requires at least one stated required_action" in error for error in errors))
self.assertTrue(any("requires at least one release condition" in error for error in errors))

def test_conditional_release_cannot_claim_unresolved_blockers(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["blockers"] = ["Resolve the fictional security boundary"]

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("incompatible with unresolved blockers" in error for error in errors))

def test_release_cannot_ignore_high_severity_finding(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["decision"] = "release"
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["release_status"] = "approved"
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("high-severity findings" in error for error in errors))

def test_critical_finding_requires_do_not_release(self) -> None:
report = copy.deepcopy(self.report)
report["findings"]["failures"][0]["severity"] = "critical"

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("critical findings require" in error for error in errors))

def test_do_not_release_can_be_supported_by_critical_finding(self) -> None:
report = copy.deepcopy(self.report)
report["findings"]["failures"][0]["severity"] = "critical"
report["recommendation"] = {
"decision": "do_not_release",
"reason": "A fictional critical boundary failure remains unresolved.",
}
report["decision"]["release_status"] = "not_approved"
report["decision"]["conditions"] = []

self.assertEqual(VALIDATOR.semantic_errors(report), [])


if __name__ == "__main__":
unittest.main()
40 changes: 28 additions & 12 deletions tools/validate_evaluation_report.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,7 +3,7 @@

The JSON Schema checks field types and allowed values. This utility adds the
cross-field rules needed to keep a recommendation, release status, conditions,
and high-severity findings internally coherent.
required actions, decision blockers, and material findings internally coherent.
"""

from __future__ import annotations
Expand DownExpand Up@@ -50,7 +50,13 @@ def schema_errors(schema: dict[str, Any], report: dict[str, Any]) -> list[str]:


def semantic_errors(report: dict[str, Any]) -> list[str]:
"""Return cross-field consistency errors after schema validation succeeds."""
"""Return cross-field consistency errors after schema validation succeeds.

A *severity* states the consequence of a finding. A *blocker* is a decision
status: unresolved blockers prevent the current release decision. Required
actions may exist on a conditional approval, but they are not blockers once
the accountable owners have accepted the bounded conditions.
"""
recommendation = report["recommendation"]
decision = report["decision"]
recommendation_decision = recommendation["decision"]
Expand All@@ -67,30 +73,40 @@ def semantic_errors(report: dict[str, Any]) -> list[str]:
)

blockers = recommendation.get("blockers", [])
required_actions = recommendation.get("required_actions", [])
conditions = decision["conditions"]
severities = [failure["severity"] for failure in report["findings"]["failures"]]
has_high = "high" in severities
has_critical = "critical" in severities

if recommendation_decision == "release":
if blockers:
errors.append("recommendation.decision=release must not include unresolved blockers")
if required_actions:
errors.append("recommendation.decision=release must not include unresolved required_actions")
if conditions:
errors.append("decision.release_status=approved must not include release conditions")

if recommendation_decision == "release_with_conditions":
if not blockers:
errors.append("release_with_conditions requires at least one stated blocker or required action")
if blockers:
errors.append(
"release_with_conditions is incompatible with unresolved blockers; "
"use required_actions and decision.conditions for bounded follow-up"
)
if not required_actions and not conditions:
errors.append(
"release_with_conditions requires at least one stated required_action or release condition"
)
if not conditions:
errors.append("approved_with_conditions requires at least one release condition")

if recommendation_decision == "do_not_release" and not blockers:
errors.append("do_not_release requires at least one stated blocker")
if recommendation_decision == "do_not_release" and not blockers and not has_critical:
errors.append("do_not_release requires at least one stated blocker or critical finding")

high_findings = [
failure
for failure in report["findings"]["failures"]
if failure["severity"] == "high"
]
if high_findings and recommendation_decision == "release":
if has_high and recommendation_decision == "release":
errors.append("recommendation.decision=release is incompatible with unresolved high-severity findings")
if has_critical and recommendation_decision != "do_not_release":
errors.append("critical findings require recommendation.decision=do_not_release")

return errors

Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 28 additions & 0 deletions docs/decision-semantics.md
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,28 @@
# Evaluation Decision Semantics

This framework keeps three concepts separate:

| Concept | Meaning | Where it is recorded |
|---|---|---|
| System risk tier | Overall impact context for the evaluated system and use case | `metadata.risk_tier` (`low`, `medium`, `high`) |
| Scenario or finding severity | Consequence if one specific scenario or failure occurs | `scenarios[].risk` and `findings.failures[].severity` (`low`, `medium`, `high`, `critical`) |
| Decision blocker | An unresolved condition that prevents the current release decision | `recommendation.blockers` |

A high- or critical-severity scenario does not automatically define the whole system's risk tier. Conversely, a high-tier system may include routine lower-severity test scenarios.

## Recommendation fields

- Use `recommendation.blockers` only for unresolved items that prevent the current decision.
- Use `recommendation.required_actions` for follow-up work accepted as part of a bounded conditional approval.
- Use `decision.conditions` for the named, accountable constraints of that conditional approval.

A report with `release_with_conditions` must not claim unresolved blockers. It must instead state the required actions and release conditions that named owners have accepted.

## Validator rules

- A plain `release` cannot include blockers, required actions, or conditions.
- A high-severity finding prevents an unconditional `release`.
- A critical finding requires `do_not_release`.
- `do_not_release` requires either a stated blocker or at least one critical finding.

These rules make report logic internally coherent. They are practitioner controls, not a substitute for organization-specific release authority, legal review, safety analysis, or compliance decisions.
6 changes: 3 additions & 3 deletions examples/sample-evaluation-report.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -10,8 +10,8 @@
},
"recommendation": {
"decision": "release_with_conditions",
"reason": "The agent met core task-completion and routing-quality targets, but failed one high-risk escalation scenario involving sensitive complaint handling.",
"blockers": [
"reason": "The agent met core task-completion and routing-quality targets, but failed one high-severity escalation scenario involving sensitive complaint handling.",
"required_actions": [
"Add deterministic escalation rule for personal-data and legal-threat complaints",
"Re-run high-risk escalation scenario pack before production release"
]
Expand DownExpand Up@@ -167,4 +167,4 @@
}
]
}
}
}
11 changes: 8 additions & 3 deletions schemas/evaluation-report.schema.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -54,6 +54,11 @@
"type": "array",
"items": { "type": "string", "minLength": 5, "maxLength": 300 },
"uniqueItems": true
},
"required_actions": {
"type": "array",
"items": { "type": "string", "minLength": 5, "maxLength": 300 },
"uniqueItems": true
}
}
},
Expand DownExpand Up@@ -83,7 +88,7 @@
"properties": {
"id": { "type": "string", "pattern": "^S-\\d{3}$" },
"scenario": { "type": "string", "minLength": 5, "maxLength": 250 },
"risk": { "type": "string", "enum": ["low", "medium", "high"] },
"risk": { "type": "string", "enum": ["low", "medium", "high", "critical"] },
"expected_behavior": { "type": "string", "minLength": 5, "maxLength": 400 },
"result": { "type": "string", "enum": ["pass", "fail", "partial"] },
"notes": { "type": "string", "maxLength": 500 }
Expand DownExpand Up@@ -123,7 +128,7 @@
"required": ["finding", "severity", "evidence", "required_action"],
"properties": {
"finding": { "type": "string", "minLength": 5, "maxLength": 300 },
"severity": { "type": "string", "enum": ["low", "medium", "high"] },
"severity": { "type": "string", "enum": ["low", "medium", "high", "critical"] },
"evidence": { "type": "string", "minLength": 2, "maxLength": 120 },
"required_action": { "type": "string", "minLength": 5, "maxLength": 300 }
}
Expand DownExpand Up@@ -175,4 +180,4 @@
}
}
}
}
}
38 changes: 33 additions & 5 deletions tests/test_validate_evaluation_report.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -27,35 +27,63 @@ def test_sample_report_is_semantically_coherent(self) -> None:
def test_release_cannot_claim_not_approved_status(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["decision"] = "release"
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["release_status"] = "not_approved"
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("requires decision.release_status=approved" in error for error in errors))

def test_release_with_conditions_requires_conditions_and_blockers(self) -> None:
def test_conditional_release_requires_conditions_or_required_actions(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("requires at least one stated blocker" in error for error in errors))
self.assertTrue(any("requires at least one stated required_action" in error for error in errors))
self.assertTrue(any("requires at least one release condition" in error for error in errors))

def test_conditional_release_cannot_claim_unresolved_blockers(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["blockers"] = ["Resolve the fictional security boundary"]

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("incompatible with unresolved blockers" in error for error in errors))

def test_release_cannot_ignore_high_severity_finding(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["decision"] = "release"
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["release_status"] = "approved"
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("high-severity findings" in error for error in errors))

def test_critical_finding_requires_do_not_release(self) -> None:
report = copy.deepcopy(self.report)
report["findings"]["failures"][0]["severity"] = "critical"

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("critical findings require" in error for error in errors))

def test_do_not_release_can_be_supported_by_critical_finding(self) -> None:
report = copy.deepcopy(self.report)
report["findings"]["failures"][0]["severity"] = "critical"
report["recommendation"] = {
"decision": "do_not_release",
"reason": "A fictional critical boundary failure remains unresolved.",
}
report["decision"]["release_status"] = "not_approved"
report["decision"]["conditions"] = []

self.assertEqual(VALIDATOR.semantic_errors(report), [])


if __name__ == "__main__":
unittest.main()
40 changes: 28 additions & 12 deletions tools/validate_evaluation_report.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,7 +3,7 @@

The JSON Schema checks field types and allowed values. This utility adds the
cross-field rules needed to keep a recommendation, release status, conditions,
and high-severity findings internally coherent.
required actions, decision blockers, and material findings internally coherent.
"""

from __future__ import annotations
Expand DownExpand Up@@ -50,7 +50,13 @@ def schema_errors(schema: dict[str, Any], report: dict[str, Any]) -> list[str]:


def semantic_errors(report: dict[str, Any]) -> list[str]:
"""Return cross-field consistency errors after schema validation succeeds."""
"""Return cross-field consistency errors after schema validation succeeds.

A *severity* states the consequence of a finding. A *blocker* is a decision
status: unresolved blockers prevent the current release decision. Required
actions may exist on a conditional approval, but they are not blockers once
the accountable owners have accepted the bounded conditions.
"""
recommendation = report["recommendation"]
decision = report["decision"]
recommendation_decision = recommendation["decision"]
Expand All@@ -67,30 +73,40 @@ def semantic_errors(report: dict[str, Any]) -> list[str]:
)

blockers = recommendation.get("blockers", [])
required_actions = recommendation.get("required_actions", [])
conditions = decision["conditions"]
severities = [failure["severity"] for failure in report["findings"]["failures"]]
has_high = "high" in severities
has_critical = "critical" in severities

if recommendation_decision == "release":
if blockers:
errors.append("recommendation.decision=release must not include unresolved blockers")
if required_actions:
errors.append("recommendation.decision=release must not include unresolved required_actions")
if conditions:
errors.append("decision.release_status=approved must not include release conditions")

if recommendation_decision == "release_with_conditions":
if not blockers:
errors.append("release_with_conditions requires at least one stated blocker or required action")
if blockers:
errors.append(
"release_with_conditions is incompatible with unresolved blockers; "
"use required_actions and decision.conditions for bounded follow-up"
)
if not required_actions and not conditions:
errors.append(
"release_with_conditions requires at least one stated required_action or release condition"
)
if not conditions:
errors.append("approved_with_conditions requires at least one release condition")

if recommendation_decision == "do_not_release" and not blockers:
errors.append("do_not_release requires at least one stated blocker")
if recommendation_decision == "do_not_release" and not blockers and not has_critical:
errors.append("do_not_release requires at least one stated blocker or critical finding")

high_findings = [
failure
for failure in report["findings"]["failures"]
if failure["severity"] == "high"
]
if high_findings and recommendation_decision == "release":
if has_high and recommendation_decision == "release":
errors.append("recommendation.decision=release is incompatible with unresolved high-severity findings")
if has_critical and recommendation_decision != "do_not_release":
errors.append("critical findings require recommendation.decision=do_not_release")

return errors

Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 28 additions & 0 deletions docs/decision-semantics.md
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,28 @@
# Evaluation Decision Semantics

This framework keeps three concepts separate:

| Concept | Meaning | Where it is recorded |
|---|---|---|
| System risk tier | Overall impact context for the evaluated system and use case | `metadata.risk_tier` (`low`, `medium`, `high`) |
| Scenario or finding severity | Consequence if one specific scenario or failure occurs | `scenarios[].risk` and `findings.failures[].severity` (`low`, `medium`, `high`, `critical`) |
| Decision blocker | An unresolved condition that prevents the current release decision | `recommendation.blockers` |

A high- or critical-severity scenario does not automatically define the whole system's risk tier. Conversely, a high-tier system may include routine lower-severity test scenarios.

## Recommendation fields

- Use `recommendation.blockers` only for unresolved items that prevent the current decision.
- Use `recommendation.required_actions` for follow-up work accepted as part of a bounded conditional approval.
- Use `decision.conditions` for the named, accountable constraints of that conditional approval.

A report with `release_with_conditions` must not claim unresolved blockers. It must instead state the required actions and release conditions that named owners have accepted.

## Validator rules

- A plain `release` cannot include blockers, required actions, or conditions.
- A high-severity finding prevents an unconditional `release`.
- A critical finding requires `do_not_release`.
- `do_not_release` requires either a stated blocker or at least one critical finding.

These rules make report logic internally coherent. They are practitioner controls, not a substitute for organization-specific release authority, legal review, safety analysis, or compliance decisions.
6 changes: 3 additions & 3 deletions examples/sample-evaluation-report.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -10,8 +10,8 @@
},
"recommendation": {
"decision": "release_with_conditions",
"reason": "The agent met core task-completion and routing-quality targets, but failed one high-risk escalation scenario involving sensitive complaint handling.",
"blockers": [
"reason": "The agent met core task-completion and routing-quality targets, but failed one high-severity escalation scenario involving sensitive complaint handling.",
"required_actions": [
"Add deterministic escalation rule for personal-data and legal-threat complaints",
"Re-run high-risk escalation scenario pack before production release"
]
Expand DownExpand Up@@ -167,4 +167,4 @@
}
]
}
}
}
11 changes: 8 additions & 3 deletions schemas/evaluation-report.schema.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -54,6 +54,11 @@
"type": "array",
"items": { "type": "string", "minLength": 5, "maxLength": 300 },
"uniqueItems": true
},
"required_actions": {
"type": "array",
"items": { "type": "string", "minLength": 5, "maxLength": 300 },
"uniqueItems": true
}
}
},
Expand DownExpand Up@@ -83,7 +88,7 @@
"properties": {
"id": { "type": "string", "pattern": "^S-\\d{3}$" },
"scenario": { "type": "string", "minLength": 5, "maxLength": 250 },
"risk": { "type": "string", "enum": ["low", "medium", "high"] },
"risk": { "type": "string", "enum": ["low", "medium", "high", "critical"] },
"expected_behavior": { "type": "string", "minLength": 5, "maxLength": 400 },
"result": { "type": "string", "enum": ["pass", "fail", "partial"] },
"notes": { "type": "string", "maxLength": 500 }
Expand DownExpand Up@@ -123,7 +128,7 @@
"required": ["finding", "severity", "evidence", "required_action"],
"properties": {
"finding": { "type": "string", "minLength": 5, "maxLength": 300 },
"severity": { "type": "string", "enum": ["low", "medium", "high"] },
"severity": { "type": "string", "enum": ["low", "medium", "high", "critical"] },
"evidence": { "type": "string", "minLength": 2, "maxLength": 120 },
"required_action": { "type": "string", "minLength": 5, "maxLength": 300 }
}
Expand DownExpand Up@@ -175,4 +180,4 @@
}
}
}
}
}
38 changes: 33 additions & 5 deletions tests/test_validate_evaluation_report.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -27,35 +27,63 @@ def test_sample_report_is_semantically_coherent(self) -> None:
def test_release_cannot_claim_not_approved_status(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["decision"] = "release"
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["release_status"] = "not_approved"
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("requires decision.release_status=approved" in error for error in errors))

def test_release_with_conditions_requires_conditions_and_blockers(self) -> None:
def test_conditional_release_requires_conditions_or_required_actions(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("requires at least one stated blocker" in error for error in errors))
self.assertTrue(any("requires at least one stated required_action" in error for error in errors))
self.assertTrue(any("requires at least one release condition" in error for error in errors))

def test_conditional_release_cannot_claim_unresolved_blockers(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["blockers"] = ["Resolve the fictional security boundary"]

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("incompatible with unresolved blockers" in error for error in errors))

def test_release_cannot_ignore_high_severity_finding(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["decision"] = "release"
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["release_status"] = "approved"
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("high-severity findings" in error for error in errors))

def test_critical_finding_requires_do_not_release(self) -> None:
report = copy.deepcopy(self.report)
report["findings"]["failures"][0]["severity"] = "critical"

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("critical findings require" in error for error in errors))

def test_do_not_release_can_be_supported_by_critical_finding(self) -> None:
report = copy.deepcopy(self.report)
report["findings"]["failures"][0]["severity"] = "critical"
report["recommendation"] = {
"decision": "do_not_release",
"reason": "A fictional critical boundary failure remains unresolved.",
}
report["decision"]["release_status"] = "not_approved"
report["decision"]["conditions"] = []

self.assertEqual(VALIDATOR.semantic_errors(report), [])


if __name__ == "__main__":
unittest.main()
40 changes: 28 additions & 12 deletions tools/validate_evaluation_report.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,7 +3,7 @@

The JSON Schema checks field types and allowed values. This utility adds the
cross-field rules needed to keep a recommendation, release status, conditions,
and high-severity findings internally coherent.
required actions, decision blockers, and material findings internally coherent.
"""

from __future__ import annotations
Expand DownExpand Up@@ -50,7 +50,13 @@ def schema_errors(schema: dict[str, Any], report: dict[str, Any]) -> list[str]:


def semantic_errors(report: dict[str, Any]) -> list[str]:
"""Return cross-field consistency errors after schema validation succeeds."""
"""Return cross-field consistency errors after schema validation succeeds.

A *severity* states the consequence of a finding. A *blocker* is a decision
status: unresolved blockers prevent the current release decision. Required
actions may exist on a conditional approval, but they are not blockers once
the accountable owners have accepted the bounded conditions.
"""
recommendation = report["recommendation"]
decision = report["decision"]
recommendation_decision = recommendation["decision"]
Expand All@@ -67,30 +73,40 @@ def semantic_errors(report: dict[str, Any]) -> list[str]:
)

blockers = recommendation.get("blockers", [])
required_actions = recommendation.get("required_actions", [])
conditions = decision["conditions"]
severities = [failure["severity"] for failure in report["findings"]["failures"]]
has_high = "high" in severities
has_critical = "critical" in severities

if recommendation_decision == "release":
if blockers:
errors.append("recommendation.decision=release must not include unresolved blockers")
if required_actions:
errors.append("recommendation.decision=release must not include unresolved required_actions")
if conditions:
errors.append("decision.release_status=approved must not include release conditions")

if recommendation_decision == "release_with_conditions":
if not blockers:
errors.append("release_with_conditions requires at least one stated blocker or required action")
if blockers:
errors.append(
"release_with_conditions is incompatible with unresolved blockers; "
"use required_actions and decision.conditions for bounded follow-up"
)
if not required_actions and not conditions:
errors.append(
"release_with_conditions requires at least one stated required_action or release condition"
)
if not conditions:
errors.append("approved_with_conditions requires at least one release condition")

if recommendation_decision == "do_not_release" and not blockers:
errors.append("do_not_release requires at least one stated blocker")
if recommendation_decision == "do_not_release" and not blockers and not has_critical:
errors.append("do_not_release requires at least one stated blocker or critical finding")

high_findings = [
failure
for failure in report["findings"]["failures"]
if failure["severity"] == "high"
]
if high_findings and recommendation_decision == "release":
if has_high and recommendation_decision == "release":
errors.append("recommendation.decision=release is incompatible with unresolved high-severity findings")
if has_critical and recommendation_decision != "do_not_release":
errors.append("critical findings require recommendation.decision=do_not_release")

return errors

Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 28 additions & 0 deletions docs/decision-semantics.md
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,28 @@
# Evaluation Decision Semantics

This framework keeps three concepts separate:

| Concept | Meaning | Where it is recorded |
|---|---|---|
| System risk tier | Overall impact context for the evaluated system and use case | `metadata.risk_tier` (`low`, `medium`, `high`) |
| Scenario or finding severity | Consequence if one specific scenario or failure occurs | `scenarios[].risk` and `findings.failures[].severity` (`low`, `medium`, `high`, `critical`) |
| Decision blocker | An unresolved condition that prevents the current release decision | `recommendation.blockers` |

A high- or critical-severity scenario does not automatically define the whole system's risk tier. Conversely, a high-tier system may include routine lower-severity test scenarios.

## Recommendation fields

- Use `recommendation.blockers` only for unresolved items that prevent the current decision.
- Use `recommendation.required_actions` for follow-up work accepted as part of a bounded conditional approval.
- Use `decision.conditions` for the named, accountable constraints of that conditional approval.

A report with `release_with_conditions` must not claim unresolved blockers. It must instead state the required actions and release conditions that named owners have accepted.

## Validator rules

- A plain `release` cannot include blockers, required actions, or conditions.
- A high-severity finding prevents an unconditional `release`.
- A critical finding requires `do_not_release`.
- `do_not_release` requires either a stated blocker or at least one critical finding.

These rules make report logic internally coherent. They are practitioner controls, not a substitute for organization-specific release authority, legal review, safety analysis, or compliance decisions.
6 changes: 3 additions & 3 deletions examples/sample-evaluation-report.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -10,8 +10,8 @@
},
"recommendation": {
"decision": "release_with_conditions",
"reason": "The agent met core task-completion and routing-quality targets, but failed one high-risk escalation scenario involving sensitive complaint handling.",
"blockers": [
"reason": "The agent met core task-completion and routing-quality targets, but failed one high-severity escalation scenario involving sensitive complaint handling.",
"required_actions": [
"Add deterministic escalation rule for personal-data and legal-threat complaints",
"Re-run high-risk escalation scenario pack before production release"
]
Expand DownExpand Up@@ -167,4 +167,4 @@
}
]
}
}
}
11 changes: 8 additions & 3 deletions schemas/evaluation-report.schema.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -54,6 +54,11 @@
"type": "array",
"items": { "type": "string", "minLength": 5, "maxLength": 300 },
"uniqueItems": true
},
"required_actions": {
"type": "array",
"items": { "type": "string", "minLength": 5, "maxLength": 300 },
"uniqueItems": true
}
}
},
Expand DownExpand Up@@ -83,7 +88,7 @@
"properties": {
"id": { "type": "string", "pattern": "^S-\\d{3}$" },
"scenario": { "type": "string", "minLength": 5, "maxLength": 250 },
"risk": { "type": "string", "enum": ["low", "medium", "high"] },
"risk": { "type": "string", "enum": ["low", "medium", "high", "critical"] },
"expected_behavior": { "type": "string", "minLength": 5, "maxLength": 400 },
"result": { "type": "string", "enum": ["pass", "fail", "partial"] },
"notes": { "type": "string", "maxLength": 500 }
Expand DownExpand Up@@ -123,7 +128,7 @@
"required": ["finding", "severity", "evidence", "required_action"],
"properties": {
"finding": { "type": "string", "minLength": 5, "maxLength": 300 },
"severity": { "type": "string", "enum": ["low", "medium", "high"] },
"severity": { "type": "string", "enum": ["low", "medium", "high", "critical"] },
"evidence": { "type": "string", "minLength": 2, "maxLength": 120 },
"required_action": { "type": "string", "minLength": 5, "maxLength": 300 }
}
Expand DownExpand Up@@ -175,4 +180,4 @@
}
}
}
}
}
38 changes: 33 additions & 5 deletions tests/test_validate_evaluation_report.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -27,35 +27,63 @@ def test_sample_report_is_semantically_coherent(self) -> None:
def test_release_cannot_claim_not_approved_status(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["decision"] = "release"
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["release_status"] = "not_approved"
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("requires decision.release_status=approved" in error for error in errors))

def test_release_with_conditions_requires_conditions_and_blockers(self) -> None:
def test_conditional_release_requires_conditions_or_required_actions(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("requires at least one stated blocker" in error for error in errors))
self.assertTrue(any("requires at least one stated required_action" in error for error in errors))
self.assertTrue(any("requires at least one release condition" in error for error in errors))

def test_conditional_release_cannot_claim_unresolved_blockers(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["blockers"] = ["Resolve the fictional security boundary"]

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("incompatible with unresolved blockers" in error for error in errors))

def test_release_cannot_ignore_high_severity_finding(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["decision"] = "release"
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["release_status"] = "approved"
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("high-severity findings" in error for error in errors))

def test_critical_finding_requires_do_not_release(self) -> None:
report = copy.deepcopy(self.report)
report["findings"]["failures"][0]["severity"] = "critical"

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("critical findings require" in error for error in errors))

def test_do_not_release_can_be_supported_by_critical_finding(self) -> None:
report = copy.deepcopy(self.report)
report["findings"]["failures"][0]["severity"] = "critical"
report["recommendation"] = {
"decision": "do_not_release",
"reason": "A fictional critical boundary failure remains unresolved.",
}
report["decision"]["release_status"] = "not_approved"
report["decision"]["conditions"] = []

self.assertEqual(VALIDATOR.semantic_errors(report), [])


if __name__ == "__main__":
unittest.main()
40 changes: 28 additions & 12 deletions tools/validate_evaluation_report.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,7 +3,7 @@

The JSON Schema checks field types and allowed values. This utility adds the
cross-field rules needed to keep a recommendation, release status, conditions,
and high-severity findings internally coherent.
required actions, decision blockers, and material findings internally coherent.
"""

from __future__ import annotations
Expand DownExpand Up@@ -50,7 +50,13 @@ def schema_errors(schema: dict[str, Any], report: dict[str, Any]) -> list[str]:


def semantic_errors(report: dict[str, Any]) -> list[str]:
"""Return cross-field consistency errors after schema validation succeeds."""
"""Return cross-field consistency errors after schema validation succeeds.

A *severity* states the consequence of a finding. A *blocker* is a decision
status: unresolved blockers prevent the current release decision. Required
actions may exist on a conditional approval, but they are not blockers once
the accountable owners have accepted the bounded conditions.
"""
recommendation = report["recommendation"]
decision = report["decision"]
recommendation_decision = recommendation["decision"]
Expand All@@ -67,30 +73,40 @@ def semantic_errors(report: dict[str, Any]) -> list[str]:
)

blockers = recommendation.get("blockers", [])
required_actions = recommendation.get("required_actions", [])
conditions = decision["conditions"]
severities = [failure["severity"] for failure in report["findings"]["failures"]]
has_high = "high" in severities
has_critical = "critical" in severities

if recommendation_decision == "release":
if blockers:
errors.append("recommendation.decision=release must not include unresolved blockers")
if required_actions:
errors.append("recommendation.decision=release must not include unresolved required_actions")
if conditions:
errors.append("decision.release_status=approved must not include release conditions")

if recommendation_decision == "release_with_conditions":
if not blockers:
errors.append("release_with_conditions requires at least one stated blocker or required action")
if blockers:
errors.append(
"release_with_conditions is incompatible with unresolved blockers; "
"use required_actions and decision.conditions for bounded follow-up"
)
if not required_actions and not conditions:
errors.append(
"release_with_conditions requires at least one stated required_action or release condition"
)
if not conditions:
errors.append("approved_with_conditions requires at least one release condition")

if recommendation_decision == "do_not_release" and not blockers:
errors.append("do_not_release requires at least one stated blocker")
if recommendation_decision == "do_not_release" and not blockers and not has_critical:
errors.append("do_not_release requires at least one stated blocker or critical finding")

high_findings = [
failure
for failure in report["findings"]["failures"]
if failure["severity"] == "high"
]
if high_findings and recommendation_decision == "release":
if has_high and recommendation_decision == "release":
errors.append("recommendation.decision=release is incompatible with unresolved high-severity findings")
if has_critical and recommendation_decision != "do_not_release":
errors.append("critical findings require recommendation.decision=do_not_release")

return errors

Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
28 changes: 28 additions & 0 deletions docs/decision-semantics.md
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,28 @@
# Evaluation Decision Semantics

This framework keeps three concepts separate:

| Concept | Meaning | Where it is recorded |
|---|---|---|
| System risk tier | Overall impact context for the evaluated system and use case | `metadata.risk_tier` (`low`, `medium`, `high`) |
| Scenario or finding severity | Consequence if one specific scenario or failure occurs | `scenarios[].risk` and `findings.failures[].severity` (`low`, `medium`, `high`, `critical`) |
| Decision blocker | An unresolved condition that prevents the current release decision | `recommendation.blockers` |

A high- or critical-severity scenario does not automatically define the whole system's risk tier. Conversely, a high-tier system may include routine lower-severity test scenarios.

## Recommendation fields

- Use `recommendation.blockers` only for unresolved items that prevent the current decision.
- Use `recommendation.required_actions` for follow-up work accepted as part of a bounded conditional approval.
- Use `decision.conditions` for the named, accountable constraints of that conditional approval.

A report with `release_with_conditions` must not claim unresolved blockers. It must instead state the required actions and release conditions that named owners have accepted.

## Validator rules

- A plain `release` cannot include blockers, required actions, or conditions.
- A high-severity finding prevents an unconditional `release`.
- A critical finding requires `do_not_release`.
- `do_not_release` requires either a stated blocker or at least one critical finding.

These rules make report logic internally coherent. They are practitioner controls, not a substitute for organization-specific release authority, legal review, safety analysis, or compliance decisions.
6 changes: 3 additions & 3 deletions examples/sample-evaluation-report.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -10,8 +10,8 @@
},
"recommendation": {
"decision": "release_with_conditions",
"reason": "The agent met core task-completion and routing-quality targets, but failed one high-risk escalation scenario involving sensitive complaint handling.",
"blockers": [
"reason": "The agent met core task-completion and routing-quality targets, but failed one high-severity escalation scenario involving sensitive complaint handling.",
"required_actions": [
"Add deterministic escalation rule for personal-data and legal-threat complaints",
"Re-run high-risk escalation scenario pack before production release"
]
Expand DownExpand Up@@ -167,4 +167,4 @@
}
]
}
}
}
11 changes: 8 additions & 3 deletions schemas/evaluation-report.schema.json
Original file line numberDiff line numberDiff line change
Expand Up@@ -54,6 +54,11 @@
"type": "array",
"items": { "type": "string", "minLength": 5, "maxLength": 300 },
"uniqueItems": true
},
"required_actions": {
"type": "array",
"items": { "type": "string", "minLength": 5, "maxLength": 300 },
"uniqueItems": true
}
}
},
Expand DownExpand Up@@ -83,7 +88,7 @@
"properties": {
"id": { "type": "string", "pattern": "^S-\\d{3}$" },
"scenario": { "type": "string", "minLength": 5, "maxLength": 250 },
"risk": { "type": "string", "enum": ["low", "medium", "high"] },
"risk": { "type": "string", "enum": ["low", "medium", "high", "critical"] },
"expected_behavior": { "type": "string", "minLength": 5, "maxLength": 400 },
"result": { "type": "string", "enum": ["pass", "fail", "partial"] },
"notes": { "type": "string", "maxLength": 500 }
Expand DownExpand Up@@ -123,7 +128,7 @@
"required": ["finding", "severity", "evidence", "required_action"],
"properties": {
"finding": { "type": "string", "minLength": 5, "maxLength": 300 },
"severity": { "type": "string", "enum": ["low", "medium", "high"] },
"severity": { "type": "string", "enum": ["low", "medium", "high", "critical"] },
"evidence": { "type": "string", "minLength": 2, "maxLength": 120 },
"required_action": { "type": "string", "minLength": 5, "maxLength": 300 }
}
Expand DownExpand Up@@ -175,4 +180,4 @@
}
}
}
}
}
38 changes: 33 additions & 5 deletions tests/test_validate_evaluation_report.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -27,35 +27,63 @@ def test_sample_report_is_semantically_coherent(self) -> None:
def test_release_cannot_claim_not_approved_status(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["decision"] = "release"
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["release_status"] = "not_approved"
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("requires decision.release_status=approved" in error for error in errors))

def test_release_with_conditions_requires_conditions_and_blockers(self) -> None:
def test_conditional_release_requires_conditions_or_required_actions(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("requires at least one stated blocker" in error for error in errors))
self.assertTrue(any("requires at least one stated required_action" in error for error in errors))
self.assertTrue(any("requires at least one release condition" in error for error in errors))

def test_conditional_release_cannot_claim_unresolved_blockers(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["blockers"] = ["Resolve the fictional security boundary"]

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("incompatible with unresolved blockers" in error for error in errors))

def test_release_cannot_ignore_high_severity_finding(self) -> None:
report = copy.deepcopy(self.report)
report["recommendation"]["decision"] = "release"
report["recommendation"].pop("blockers")
report["recommendation"].pop("required_actions")
report["decision"]["release_status"] = "approved"
report["decision"]["conditions"] = []

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("high-severity findings" in error for error in errors))

def test_critical_finding_requires_do_not_release(self) -> None:
report = copy.deepcopy(self.report)
report["findings"]["failures"][0]["severity"] = "critical"

errors = VALIDATOR.semantic_errors(report)

self.assertTrue(any("critical findings require" in error for error in errors))

def test_do_not_release_can_be_supported_by_critical_finding(self) -> None:
report = copy.deepcopy(self.report)
report["findings"]["failures"][0]["severity"] = "critical"
report["recommendation"] = {
"decision": "do_not_release",
"reason": "A fictional critical boundary failure remains unresolved.",
}
report["decision"]["release_status"] = "not_approved"
report["decision"]["conditions"] = []

self.assertEqual(VALIDATOR.semantic_errors(report), [])


if __name__ == "__main__":
unittest.main()
40 changes: 28 additions & 12 deletions tools/validate_evaluation_report.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -3,7 +3,7 @@

The JSON Schema checks field types and allowed values. This utility adds the
cross-field rules needed to keep a recommendation, release status, conditions,
and high-severity findings internally coherent.
required actions, decision blockers, and material findings internally coherent.
"""

from __future__ import annotations
Expand DownExpand Up@@ -50,7 +50,13 @@ def schema_errors(schema: dict[str, Any], report: dict[str, Any]) -> list[str]:


def semantic_errors(report: dict[str, Any]) -> list[str]:
"""Return cross-field consistency errors after schema validation succeeds."""
"""Return cross-field consistency errors after schema validation succeeds.

A *severity* states the consequence of a finding. A *blocker* is a decision
status: unresolved blockers prevent the current release decision. Required
actions may exist on a conditional approval, but they are not blockers once
the accountable owners have accepted the bounded conditions.
"""
recommendation = report["recommendation"]
decision = report["decision"]
recommendation_decision = recommendation["decision"]
Expand All@@ -67,30 +73,40 @@ def semantic_errors(report: dict[str, Any]) -> list[str]:
)

blockers = recommendation.get("blockers", [])
required_actions = recommendation.get("required_actions", [])
conditions = decision["conditions"]
severities = [failure["severity"] for failure in report["findings"]["failures"]]
has_high = "high" in severities
has_critical = "critical" in severities

if recommendation_decision == "release":
if blockers:
errors.append("recommendation.decision=release must not include unresolved blockers")
if required_actions:
errors.append("recommendation.decision=release must not include unresolved required_actions")
if conditions:
errors.append("decision.release_status=approved must not include release conditions")

if recommendation_decision == "release_with_conditions":
if not blockers:
errors.append("release_with_conditions requires at least one stated blocker or required action")
if blockers:
errors.append(
"release_with_conditions is incompatible with unresolved blockers; "
"use required_actions and decision.conditions for bounded follow-up"
)
if not required_actions and not conditions:
errors.append(
"release_with_conditions requires at least one stated required_action or release condition"
)
if not conditions:
errors.append("approved_with_conditions requires at least one release condition")

if recommendation_decision == "do_not_release" and not blockers:
errors.append("do_not_release requires at least one stated blocker")
if recommendation_decision == "do_not_release" and not blockers and not has_critical:
errors.append("do_not_release requires at least one stated blocker or critical finding")

high_findings = [
failure
for failure in report["findings"]["failures"]
if failure["severity"] == "high"
]
if high_findings and recommendation_decision == "release":
if has_high and recommendation_decision == "release":
errors.append("recommendation.decision=release is incompatible with unresolved high-severity findings")
if has_critical and recommendation_decision != "do_not_release":
errors.append("critical findings require recommendation.decision=do_not_release")

return errors

Expand Down
Loading