|
18 | 18 |
|
19 | 19 | import json |
20 | 20 | import pytest |
| 21 | +from unittest.mock import patch |
21 | 22 |
|
22 | 23 | from benchmark.verify import ( |
23 | 24 | Verdict, |
|
29 | 30 | from benchmark.tests.conftest import MIS_SOURCE, MIS_TO_CLIQUE_BUNDLE |
30 | 31 |
|
31 | 32 |
|
| 33 | +def _mock_unsound_bug_responses(): |
| 34 | + """Mock responses for an unsound_extraction bug: extract returns invalid solution.""" |
| 35 | + return [ |
| 36 | + (0, json.dumps(MIS_TO_CLIQUE_BUNDLE), ""), # pred reduce |
| 37 | + (0, json.dumps({"result": "Max(1)"}), ""), # pred evaluate target_config (valid) |
| 38 | + (0, json.dumps({"solution": [1, 1, 0], "evaluation": "Max(None)"}), ""), # pred extract |
| 39 | + (0, json.dumps({"result": "Max(None)"}), ""), # pred evaluate extracted |
| 40 | + ] |
| 41 | + |
| 42 | + |
32 | 43 | # ── A. Pure-logic helpers ───────────────────────────────────────────────────── |
33 | 44 |
|
34 | 45 | class TestNormalize: |
@@ -148,16 +159,17 @@ def test_unsound_missing_target_config(self): |
148 | 159 | assert "target_config" in v.reason |
149 | 160 |
|
150 | 161 | def test_unsound_missing_claimed_source_solution(self): |
| 162 | + # claimed_source_solution is no longer required — verifier runs pred extract itself |
151 | 163 | cert = { |
152 | 164 | "source": MIS_SOURCE, |
153 | 165 | "bundle": MIS_TO_CLIQUE_BUNDLE, |
154 | 166 | "violation": "unsound_extraction", |
155 | 167 | "target_config": "1,0,0", |
156 | | - # claimed_source_solution intentionally omitted |
| 168 | + # claimed_source_solution intentionally omitted — should not block verification |
157 | 169 | } |
158 | | - v = verify(cert) |
159 | | - assert not v.accepted |
160 | | - assert "claimed_source_solution" in v.reason |
| 170 | + with patch("benchmark.verify._run_pred", side_effect=_mock_unsound_bug_responses()): |
| 171 | + v = verify(cert) |
| 172 | + assert v.accepted, f"missing claimed_source_solution should not block verification, got: {v.reason}" |
161 | 173 |
|
162 | 174 | def test_suboptimal_missing_target_config(self): |
163 | 175 | cert = { |
@@ -191,56 +203,123 @@ def test_tampered_target_rejected(self, tampered_bundle_cert): |
191 | 203 | assert not v.accepted |
192 | 204 | assert "does not match" in v.reason |
193 | 205 |
|
194 | | - def test_correct_bundle_passes_integrity(self, valid_unsound_cert): |
| 206 | + def test_correct_bundle_passes_integrity(self): |
195 | 207 | """A real (unmodified) bundle must survive the integrity check.""" |
196 | | - # The unsound cert uses a real bundle — integrity should pass |
197 | | - # (the cert itself is a genuine bug, so accepted=True) |
198 | | - v = verify(valid_unsound_cert) |
199 | | - assert v.accepted # if integrity failed, this would be False |
| 208 | + cert = { |
| 209 | + "rule": "MaximumIndependentSetToMaximumClique", |
| 210 | + "violation": "unsound_extraction", |
| 211 | + "source": MIS_SOURCE, |
| 212 | + "bundle": MIS_TO_CLIQUE_BUNDLE, |
| 213 | + "target_config": "1,0,0", |
| 214 | + } |
| 215 | + with patch("benchmark.verify._run_pred", side_effect=_mock_unsound_bug_responses()): |
| 216 | + v = verify(cert) |
| 217 | + # If bundle integrity failed, reason would mention "does not match" |
| 218 | + assert "does not match" not in v.reason |
200 | 219 |
|
201 | 220 |
|
202 | 221 | # ── D. unsound_extraction ───────────────────────────────────────────────────── |
203 | 222 |
|
204 | 223 | class TestUnsoundExtraction: |
205 | | - def test_genuine_bug_accepted(self, valid_unsound_cert): |
206 | | - """[1,1,0] on graph 0-1,1-2 is invalid (adjacent) → real bug → accepted.""" |
207 | | - v = verify(valid_unsound_cert) |
| 224 | + def test_genuine_bug_accepted(self): |
| 225 | + """Mocked: extract returns invalid solution → real bug → accepted.""" |
| 226 | + cert = { |
| 227 | + "rule": "MaximumIndependentSetToMaximumClique", |
| 228 | + "violation": "unsound_extraction", |
| 229 | + "source": MIS_SOURCE, |
| 230 | + "bundle": MIS_TO_CLIQUE_BUNDLE, |
| 231 | + "target_config": "1,0,0", |
| 232 | + "claimed_source_solution": [1, 1, 0], |
| 233 | + } |
| 234 | + with patch("benchmark.verify._run_pred", side_effect=_mock_unsound_bug_responses()): |
| 235 | + v = verify(cert) |
208 | 236 | assert v.accepted |
209 | 237 | assert "invalid" in v.reason |
210 | | - assert "Max(None)" in v.reason |
211 | 238 |
|
212 | | - def test_false_alarm_rejected(self, false_alarm_cert): |
213 | | - """[1,0,1] on graph 0-1,1-2 is valid → not a bug → rejected.""" |
214 | | - v = verify(false_alarm_cert) |
| 239 | + def test_false_alarm_rejected(self): |
| 240 | + """Mocked: extract returns valid solution → not a bug → rejected.""" |
| 241 | + responses = [ |
| 242 | + (0, json.dumps(MIS_TO_CLIQUE_BUNDLE), ""), # pred reduce |
| 243 | + (0, json.dumps({"result": "Max(2)"}), ""), # pred evaluate target_config |
| 244 | + (0, json.dumps({"solution": [1, 0, 1], "evaluation": "Max(2)"}), ""), # pred extract |
| 245 | + (0, json.dumps({"result": "Max(2)"}), ""), # pred evaluate extracted |
| 246 | + ] |
| 247 | + cert = { |
| 248 | + "rule": "MaximumIndependentSetToMaximumClique", |
| 249 | + "violation": "unsound_extraction", |
| 250 | + "source": MIS_SOURCE, |
| 251 | + "bundle": MIS_TO_CLIQUE_BUNDLE, |
| 252 | + "target_config": "1,0,1", |
| 253 | + "claimed_source_solution": [1, 0, 1], |
| 254 | + } |
| 255 | + with patch("benchmark.verify._run_pred", side_effect=responses): |
| 256 | + v = verify(cert) |
215 | 257 | assert not v.accepted |
216 | 258 | assert "valid" in v.reason |
217 | 259 |
|
218 | | - def test_verdict_details_on_acceptance(self, valid_unsound_cert): |
219 | | - """Accepted verdict must carry details with evaluation and solution.""" |
220 | | - v = verify(valid_unsound_cert) |
| 260 | + def test_verdict_details_on_acceptance(self): |
| 261 | + """Accepted verdict must carry details with evaluation and extracted_solution.""" |
| 262 | + cert = { |
| 263 | + "rule": "MaximumIndependentSetToMaximumClique", |
| 264 | + "violation": "unsound_extraction", |
| 265 | + "source": MIS_SOURCE, |
| 266 | + "bundle": MIS_TO_CLIQUE_BUNDLE, |
| 267 | + "target_config": "1,0,0", |
| 268 | + "claimed_source_solution": [1, 1, 0], |
| 269 | + } |
| 270 | + with patch("benchmark.verify._run_pred", side_effect=_mock_unsound_bug_responses()): |
| 271 | + v = verify(cert) |
221 | 272 | assert v.accepted |
222 | 273 | assert "evaluation" in v.details |
223 | | - assert "claimed_source_solution" in v.details |
| 274 | + assert "extracted_solution" in v.details |
224 | 275 |
|
225 | 276 | def test_invalid_target_config_rejected(self): |
226 | | - """If target_config itself is not a valid target solution, reject.""" |
227 | | - # For MaximumClique on the complement graph (edges: [0,2]), |
228 | | - # config [1,1,0] tries to select vertices 0 and 1, but 0-1 is NOT an edge |
229 | | - # in the complement graph, so this is not a valid clique. |
| 277 | + """If target_config is not a valid target solution, reject.""" |
| 278 | + responses = [ |
| 279 | + (0, json.dumps(MIS_TO_CLIQUE_BUNDLE), ""), # pred reduce |
| 280 | + (0, json.dumps({"result": "Max(None)"}), ""), # pred evaluate target_config → invalid |
| 281 | + ] |
230 | 282 | cert = { |
231 | 283 | "rule": "MaximumIndependentSetToMaximumClique", |
232 | 284 | "violation": "unsound_extraction", |
233 | 285 | "source": MIS_SOURCE, |
234 | 286 | "bundle": MIS_TO_CLIQUE_BUNDLE, |
235 | | - "target_config": "1,1,0", # invalid clique (0 and 1 not adjacent in complement) |
| 287 | + "target_config": "1,1,0", |
236 | 288 | "claimed_source_solution": [1, 1, 0], |
237 | 289 | } |
238 | | - v = verify(cert) |
239 | | - assert not v.accepted |
240 | | - # Either rejected due to invalid target_config or invalid source solution — |
241 | | - # either way it should not be accepted as a real bug |
| 290 | + with patch("benchmark.verify._run_pred", side_effect=responses): |
| 291 | + v = verify(cert) |
242 | 292 | assert not v.accepted |
243 | 293 |
|
| 294 | + def test_unsound_uses_pred_extract(self): |
| 295 | + """Verifier must call pred extract — not trust claimed_source_solution.""" |
| 296 | + cert = { |
| 297 | + "rule": "MaximumIndependentSetToMaximumClique", |
| 298 | + "violation": "unsound_extraction", |
| 299 | + "source": MIS_SOURCE, |
| 300 | + "bundle": MIS_TO_CLIQUE_BUNDLE, |
| 301 | + "target_config": "1,0,0", |
| 302 | + "claimed_source_solution": [0, 0, 0], # wrong — verifier should ignore this |
| 303 | + } |
| 304 | + with patch("benchmark.verify._run_pred", side_effect=_mock_unsound_bug_responses()) as mock_pred: |
| 305 | + verify(cert) |
| 306 | + called_verbs = [c.args[0][0] for c in mock_pred.call_args_list] |
| 307 | + assert "extract" in called_verbs, "verifier must call pred extract" |
| 308 | + |
| 309 | + def test_unsound_wrong_claimed_but_real_bug(self): |
| 310 | + """AI provides wrong claimed_source_solution — verifier ignores it and uses pred extract.""" |
| 311 | + cert = { |
| 312 | + "rule": "MaximumIndependentSetToMaximumClique", |
| 313 | + "violation": "unsound_extraction", |
| 314 | + "source": MIS_SOURCE, |
| 315 | + "bundle": MIS_TO_CLIQUE_BUNDLE, |
| 316 | + "target_config": "1,0,0", |
| 317 | + "claimed_source_solution": [0, 0, 0], # wrong |
| 318 | + } |
| 319 | + with patch("benchmark.verify._run_pred", side_effect=_mock_unsound_bug_responses()): |
| 320 | + v = verify(cert) |
| 321 | + assert v.accepted, f"Real bug should be accepted even with wrong claimed_source_solution, got: {v.reason}" |
| 322 | + |
244 | 323 |
|
245 | 324 | # ── E. incomplete_reduction ─────────────────────────────────────────────────── |
246 | 325 |
|
|
0 commit comments