From 7ef4238bf7374195271eaba9c73f54e4dcb2cfbd Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E9=BE=9A=E6=81=92?= Date: Tue, 28 Jul 2026 21:04:23 +0800 Subject: [PATCH] feat(24_miracle): add reproducible matrix control plane --- .../24_miracle_evaluation_protocol.v0.3.json | 112 +++ docs/games/24_miracle_roster_manifest.json | 39 + src/agentbench_frame/games/miracle/matrix.py | 316 ++++++ .../games/miracle/matrix_runner.py | 945 ++++++++++++++++++ tests/miracle/test_atomic_windows_retry.py | 39 + tests/miracle/test_matrix.py | 223 +++++ tests/miracle/test_matrix_cli_flow.py | 247 +++++ tests/miracle/test_matrix_cli_inputs.py | 278 ++++++ tests/miracle/test_matrix_identity.py | 121 +++ tests/miracle/test_matrix_resume_assets.py | 176 ++++ tests/miracle/test_matrix_runner.py | 218 ++++ tests/miracle/test_section2_strictness.py | 620 ++++++++++++ tests/miracle/test_section4_full_chain.py | 683 +++++++++++++ tools/miracle_matrix.py | 614 ++++++++++++ 14 files changed, 4631 insertions(+) create mode 100644 docs/games/24_miracle_evaluation_protocol.v0.3.json create mode 100644 docs/games/24_miracle_roster_manifest.json create mode 100644 src/agentbench_frame/games/miracle/matrix.py create mode 100644 src/agentbench_frame/games/miracle/matrix_runner.py create mode 100644 tests/miracle/test_atomic_windows_retry.py create mode 100644 tests/miracle/test_matrix.py create mode 100644 tests/miracle/test_matrix_cli_flow.py create mode 100644 tests/miracle/test_matrix_cli_inputs.py create mode 100644 tests/miracle/test_matrix_identity.py create mode 100644 tests/miracle/test_matrix_resume_assets.py create mode 100644 tests/miracle/test_matrix_runner.py create mode 100644 tests/miracle/test_section2_strictness.py create mode 100644 tests/miracle/test_section4_full_chain.py create mode 100644 tools/miracle_matrix.py diff --git a/docs/games/24_miracle_evaluation_protocol.v0.3.json b/docs/games/24_miracle_evaluation_protocol.v0.3.json new file mode 100644 index 0000000..1024c1d --- /dev/null +++ b/docs/games/24_miracle_evaluation_protocol.v0.3.json @@ -0,0 +1,112 @@ +{ + "protocol_version": "0.3-pre-registered", + "status": "PRE_REGISTERED_NOT_AUTHORIZED", + "_doc": "24_miracle 正式评测协议预注册草案。状态 DRAFT_NOT_AUTHORIZED:未冻结、未执行、未创建正式矩阵目录或 progress 文件。仅当局人明确授权后冻结并启动。", + "_generated": "2026-07-21", + + "matrix_meaning": { + "status": "USER_DECIDED_PLAN_A", + "chosen": "A_ifelse_vs_16(用户正式选择:if-else 对 rank01–16,每对手换阵营各 1 局,共 32 局)", + "evidence": "高翔 historical lessons 明确研究目标为 'if-else bot vs 每位人类决赛选手'('Test against unchanged human finalists');战力记录均为 if-else vs rankNN。16×16 round-robin 在 SKILL.md 中被列为与 '完整16人矩阵' 不同的独立禁项。但精确协议(每对手局数/批量结构/timeout)材料未唯一确定。", + "options": [ + {"id": "A_ifelse_vs_16", "desc": "if-else Agent 分别对 16 个决赛策略,双方换 camp(最符合历史研究目标)", "estimated_games": "16 对手 × 2 camp × m 局/配置;m=1→32 局,m=5→160 局"}, + {"id": "B_round_robin_16x16", "desc": "16 策略间完整 round-robin(与 A 不同,规模大得多)", "estimated_games": "16×15=240 配对 × 换 camp × m 局;m=1→480 局,m=5→2400 局"}, + {"id": "C_other", "desc": "其他用户指定矩阵"} + ], + "do_not_conflate": true, + "total_attempts": 32, + "games_per_opponent": 2, + "camps_per_opponent": [0, 1], + "games_per_opponent_per_camp": 1 + }, + + "frozen_identities": { + "evaluated_agent": {"name": "miracle_ifelse", "source": "高翔 ifelse_bot/main.py", "sha256": "98199fae8875de63b41d2eacd92ad5c58b4d5aa95b01b05aeacedd0b55402f4b", "modifiable": false}, + "judge": {"source": "external_asset:judge_dev_logic", "main_py_sha256": "104f77bf4ec59b96b46ffc05e20482319f40fc83401a9becd785d0be98a7fc09", "modifiable": false}, + "opponents": "见 docs/games/24_miracle_roster_manifest.json(16 策略 archive_sha256)", + "build_artifacts_win64_mingw": { + "_note": "阶段9A 预检隔离编译产物 SHA256(g++ 15.2.0 MinGW, GNU Make 4.4.1,策略自带 makefile 原样编译)", + "rank01": "b451d4f99b694c4ac446749d904901b1235a1f8b7a52d4b06e738ceecc4b0347", + "rank02": "5c731a794eb8282f242defee099685fb20fbd7f2d7644046b5fba0385ad555f4", + "rank03": "84cdb1344104060e6341efea691077d0e3962ef60ad523e25126b18365d581eb", + "rank06": "b10de17f886ca930a8bad9a4136c1643780f26916018715486f6195332c3e704", + "rank08": "249b9c6f7bcea4b384089a0093eebed29fb67530a078fbf217cb7c274e8c6c4f", + "rank09": "3c7c822cec8cf613ff32210f61faa95b157225c51069e450d54ca1e39178be0f", + "rank10": "ae0c24618b1fdc3e88f64a3b5b61ab0edfd144fd5259f26aa2c40ef5d6467287", + "rank11": "6c8c169c1f23dc876969ac7f50f17aa05f3932bd51f7142d56c06f0757242037", + "rank12": "e92fe57054968b7b35f8e61a7fb2ac591e801b92f447e2043f9f0fa793c4c2ab", + "rank13": "d2e8ff65eaed3c979d1a5c2deb5a9e2cfafa7a3a8105690908bd4e08d4d647e9", + "rank14": "bb0f1b171251edc39a9b424c50207b529bb318dc0e82186551d899f24a26cc01", + "rank15": "c570d1807402b9fcfe9db841d88420c3cac8a5bbf1f9a2ec9e3a2b51cb7a6fb3", + "rank16": "ed92b36ba411cd52c1408762bbb6cedfc54a7a62ecb16f7d2d98cb95458372db", + "rank04_rank05_rank07": "Python,无需编译(main.py)", + "rank16_blocker": "已解除(阶段9B):rank16 隔离构建副本内创建空 build/(构建环境准备,非策略修复;源码/Makefile 哈希编译前后一致),原 Makefile 编译成功。rank16 main.exe SHA256 已补入。编译警告 'control reaches end of non-void function' 保留为运行风险(不修复/不隐瞒,见 known_runtime_risks)。", + "rank03_note": "编译+启动成功,但保留历史运行时崩溃风险(invalid_now);编译/启动成功 ≠ 比赛可用。" + } + }, + + "seed_policy": { + "requested_seed": null, + "effective_seed": null, + "deterministic_seed_supported": false, + "reproducible_from_seed": false, + "recorded_only": "realized_randomization={map_type, day_time}(Replay 头读出,不表述为 seed)", + "rationale": "Judge 用 random.randint 选 map_type/day_time,不读外部 seed;不修改 Judge" + }, + + "timeouts": { + "per_ai_operation_s": 8.0, + "wrapper_per_game_s": 120.0, + "note": "MAX_ROUND=100;最坏单局≈ wrapper_per_game_s;if-else 局可更长(smoke g2_01=630 steps 仍在限内)" + }, + + "validity": { + "valid_game": "合法 end_info + result-json 与 trace 分数一致 + raw_winner 符合 Judge 规则 + end_info 前无 ai_error/ai_timeout + Judge 未在 end_info 前崩溃 + Replay 存在且合法 + cleanup 完成", + "invalid_game": "AI crash / AI timeout / Judge crash / wrapper timeout / evidence_mismatch / replay_missing / replay_corrupt / result_json_missing / result_json_corrupt", + "rank03": "invalid_now(对手程序崩溃),不计入有效胜率" + }, + + "classification": { + "ai_crash": "AI 在 end_info 前自然异常退出,或 trace 出现 ai_error", + "ai_timeout": "trace 出现 ai_timeout,或对应 AI 在动作期限内未响应", + "runner_cleanup_nonzero": "end_info 后 runner 主动终止 AI 产生的非零 returncode —— 视为正常 cleanup,不判 AI 崩溃", + "judge_crash": "Judge 在 end_info 前异常退出,或无合法 end_info 且 Judge 非正常退出", + "wrapper_timeout": "vendor 及 Judge/AI 树未在 wrapper_per_game_s 内结束", + "infra_failure": "result-json 缺失/损坏、证据三向矛盾、Replay 损坏等基础设施层失败" + }, + + "scoring": { + "raw_winner_rule": "end_info={'0':s0,'1':s1}; winner = 0 if s0>s1 else 1", + "tie": "score0==score1 时 Judge 判 player1 获胜(judge_tiebreak_applied=true),记为 win/loss 而非 draw", + "win_rate": "wins / valid_games(分母仅 valid games);valid_games==0 时 win_rate=null, evaluation_status=NO_VALID_GAMES(不报 0%)", + "draw": "仅用于异常输入/未来协议防御;当前 Judge 不产生 draw" + }, + + "execution": { + "no_rerun_successful": true, + "recovery": "失败恢复必须使用新 game_id 并注明 recovery_of=<原 game_id>;本轮不自动恢复", + "batching": "分批执行;每批设停止门槛(残留进程/证据矛盾/winner 映射不一致/数据损坏即停,不进入下一批)", + "camp_swap": "每对手双方各占 camp0/camp1(换边)", + "pid_createtime_residual_check": "每局后按精确 PID + psutil create_time 独立核验 judge/ai0/ai1 无残留;PID 复用不杀", + "cross_validation": "Replay 头/哈希、trace、result-json、events、summary、网页六处可相互追溯;任一权威字段冲突即 evidence_mismatch", + "failure_evidence_permanent": true, + "smoke_not_competitiveness": "4 局 smoke 仅为基础设施验证,不作为 if-else 竞争力结论" + }, + + "estimate": { + "time": "取决于 matrix_meaning 选择与 m:A 方案 32 局≈0.5–1.5h,160 局≈2.5–7h;B 方案 480 局≈4–12h,2400 局≈1–2 天(含 C++ 编译)", + "storage": "每局≈ trace+replay+result-json+stdout+stderr(KB–数十 KB 级);A 方案<50MB,B 方案可达 GB 级", + "worst_case": "每局逼近 wrapper_per_game_s;总时间≈总局数×wrapper_per_game_s" + }, + + "known_runtime_risks": [ + {"id": "rank03_crash", "strategy": "rank03", "risk": "历史运行时崩溃(SKILL.md 记载当前环境对手程序崩溃)", "handling": "保持原策略不变;正式对局若在 end_info 前崩溃则记 invalid 并保留原始证据;不补跑;不计入 Agent 有效胜率(不进 win_rate 分母)", "not_an_infra_blocker": true}, + {"id": "rank16_compiler_warning", "strategy": "rank16", "risk": "编译警告 control reaches end of non-void function(ai-sample.cpp 等)", "handling": "属运行风险,不修复、不隐瞒、不改源码;正式对局异常按 invalid 处理并保留证据"}, + {"id": "general_cpp_warnings", "risk": "多个 C++ 策略编译产生 -Wreturn-type 等警告(非错误)", "handling": "未修改任何策略源码/Makefile/优化;警告仅记录,运行时异常按分类规则处理"} + ], + "blockers_before_matrix": [ + "FORMAL_32_GAME_MATRIX_NOT_AUTHORIZED" + ], + + "not_authorized": ["完整16人矩阵执行", "16×16 round-robin", "RL/Round81/hidden eval", "修改 if-else 策略", "push/PR/上传 Results/合并 main", "删除或重跑 smoke 证据"] +} diff --git a/docs/games/24_miracle_roster_manifest.json b/docs/games/24_miracle_roster_manifest.json new file mode 100644 index 0000000..3001b26 --- /dev/null +++ b/docs/games/24_miracle_roster_manifest.json @@ -0,0 +1,39 @@ +{ + "_doc": "24_miracle 16 决赛策略运行时兼容性静态盘点。仅静态检查与轻量入口审计;未编译、未启动任何 Judge/AI 比赛/完整策略。每个策略的 verification_status 标注为 static_only,绝不写成已成功运行。", + "_generated": "2026-07-21", + "_source_extracted": "external_asset:24_miracle_final/extracted", + "_source_archives": "external_asset:24_miracle_final/archives", + "_toolchain_observed": { + "g++": "MinGW 15.2.0 (x86_64-win32-seh) available at /c/Program Files/mingw64/bin/g++", + "make": "GNU Make 4.4.1 available", + "note": "本机有 g++/make,可编译 C++ 策略;但 MinGW `g++ -o main` 产出 main.exe,而 vendor/上游 run_match.resolve_ai_command 只探测 ./main(无扩展名),故 Windows 下编译后仍无法按现 runner 协议启动 C++ 策略。" + }, + "_runner_launch_gap": "vendor/miracle_local/run_match.py::resolve_ai_command 仅识别 ./main 与 main.py;MinGW 产出的 main.exe 不被识别。C++ 策略需显式 --p0-cmd/--p1-cmd main.exe、或适配层补 main.exe 探测、或在 Linux/WSL 运行。", + "strategies": [ + {"rank": 1, "username": "Bruce", "display_name": "朱昱熹", "entity": "Maiev", "language": "make", "version": 10, "type": "cpp_source", "entry": "makefile -> main (g++ -std=c++11 gameunit.cpp calculator.cpp ai_client.cpp ai.cpp -o main)", "archive_sha256": "b4603d7197447a3df9bfff126da9340659ea6977d320814febb79a5df0df4fed", "windows_native_runnable": "not_without_compile_and_mainexe_fix", "needs_compile": true, "needs_wsl_or_linux": false, "compiled_artifacts_present": false, "missing_files": false, "runner_launchable_via_current_protocol": false, "verification_status": "static_only"}, + {"rank": 2, "username": "wiku30", "display_name": "赵梓硕", "entity": "碧海潮生曲", "language": "make", "version": 18, "type": "cpp_source", "entry": "makefile -> main (含 backup.bat,仅为 git 操作非构建)", "archive_sha256": "6b440894709982433cddbe047bae211e7b45988c3d1cfd2dd0bf4e91c38b7708", "windows_native_runnable": "not_without_compile_and_mainexe_fix", "needs_compile": true, "needs_wsl_or_linux": false, "compiled_artifacts_present": false, "missing_files": false, "runner_launchable_via_current_protocol": false, "verification_status": "static_only"}, + {"rank": 3, "username": "nzhtl1477", "display_name": "李欣隆", "entity": "深浅值藏的第六分块", "language": "make", "version": 1, "type": "cpp_source", "entry": "makefile -> main", "archive_sha256": "6c526e491ccc5090ece014ababde0c5c021acc3c789981cf50cd204262e7547e", "windows_native_runnable": "not_without_compile_and_mainexe_fix", "needs_compile": true, "needs_wsl_or_linux": false, "compiled_artifacts_present": false, "missing_files": false, "runner_launchable_via_current_protocol": false, "verification_status": "static_only", "invalid_now": true, "invalid_reason": "SKILL.md 记载当前环境对手程序崩溃,不能算 Agent 胜利"}, + {"rank": 4, "username": "SuperJasper", "display_name": "宋子萌", "entity": "K", "language": "python_zip", "version": 1, "type": "python_script", "entry": "main.py", "runnable_sha256": "1dfe69738141f80f7d2fb5ce1a8321126141360d86d515caf66d3f4010d86f88", "archive_sha256": "ec7c574c21625736ae38dede760e0cdb796b9cb375216f9086b28c00fe07d2e2", "windows_native_runnable": true, "needs_compile": false, "needs_wsl_or_linux": false, "compiled_artifacts_present": false, "missing_files": false, "runner_launchable_via_current_protocol": true, "verification_status": "static_only", "skill_note": "有效战胜(SKILL.md)"}, + {"rank": 5, "username": "zex18", "display_name": "周恩贤", "entity": "zex的人工智障", "language": "python_zip", "version": 56, "type": "python_script", "entry": "main.py", "runnable_sha256": "4166a54d042e9e2e5b37cf172705dbf7617872ffb1184adb3ceb697bcc6b9707", "archive_sha256": "363d68c5fbdf1e6d869ca3a3612fd67b76fae8abc051b3ad164a8942bed0ec87", "windows_native_runnable": true, "needs_compile": false, "needs_wsl_or_linux": false, "compiled_artifacts_present": false, "missing_files": false, "runner_launchable_via_current_protocol": true, "verification_status": "static_only", "skill_note": "较接近的强对手"}, + {"rank": 6, "username": "robinliu", "display_name": "刘子奇", "entity": "Mooncell", "language": "make", "version": 78, "type": "cpp_source", "entry": "makefile -> main", "archive_sha256": "dc25d4357ebee7aa6274f8b54cee885c5387a62e1ec62dcc71446f4309cd4806", "windows_native_runnable": "not_without_compile_and_mainexe_fix", "needs_compile": true, "needs_wsl_or_linux": false, "compiled_artifacts_present": false, "missing_files": false, "runner_launchable_via_current_protocol": false, "verification_status": "static_only"}, + {"rank": 7, "username": "blazingBonfire", "display_name": "陈英发", "entity": "守护者", "language": "python_zip", "version": 22, "type": "python_script", "entry": "main.py", "runnable_sha256": "eab339ef43fcc521ec8eb7bcb39859e35e8ea4681bb00661a8faf91c5d63816c", "archive_sha256": "410269e658407193786328505fb0bdf8b585c295ed5322315d6ac35a8c755ba1", "windows_native_runnable": true, "needs_compile": false, "needs_wsl_or_linux": false, "compiled_artifacts_present": false, "missing_files": false, "runner_launchable_via_current_protocol": true, "verification_status": "static_only", "skill_note": "接近但未解决"}, + {"rank": 8, "username": "wenyi", "display_name": "洪文逸", "entity": "yyy", "language": "make", "version": 1, "type": "cpp_source", "entry": "makefile -> main", "archive_sha256": "81e4d6a21a5790d7384be8cf9c8fe172b7fdc97933186204a05fe28e52914461", "windows_native_runnable": "not_without_compile_and_mainexe_fix", "needs_compile": true, "needs_wsl_or_linux": false, "compiled_artifacts_present": false, "missing_files": false, "runner_launchable_via_current_protocol": false, "verification_status": "static_only"}, + {"rank": 9, "username": "hsiachi", "display_name": "夏奇", "entity": "タチコマ(na)", "language": "make", "version": 8, "type": "cpp_source", "entry": "makefile -> main", "archive_sha256": "48e2db35e6804ba4e0b8717fd2fd4b50d9ea931dd33a158fbe914838b69a2701", "windows_native_runnable": "not_without_compile_and_mainexe_fix", "needs_compile": true, "needs_wsl_or_linux": false, "compiled_artifacts_present": false, "missing_files": false, "runner_launchable_via_current_protocol": false, "verification_status": "static_only"}, + {"rank": 10, "username": "SD_le", "display_name": "杨卓毅", "entity": "AK", "language": "make", "version": 8, "type": "cpp_source", "entry": "makefile -> main", "archive_sha256": "efab49207a2f89509ebf64dc286b31f1147e1af65a4fd0bedca28d66d1539f50", "windows_native_runnable": "not_without_compile_and_mainexe_fix", "needs_compile": true, "needs_wsl_or_linux": false, "compiled_artifacts_present": false, "missing_files": false, "runner_launchable_via_current_protocol": false, "verification_status": "static_only"}, + {"rank": 11, "username": "jasonvictoryan", "display_name": "颜杰龙", "entity": "Halcyon", "language": "make", "version": 14, "type": "cpp_source", "entry": "makefile -> main", "archive_sha256": "656d009766064a096f135a8aca6b6a7d2f695e5c0eed2f3b85aba7b6d3ed812e", "windows_native_runnable": "not_without_compile_and_mainexe_fix", "needs_compile": true, "needs_wsl_or_linux": false, "compiled_artifacts_present": false, "missing_files": false, "runner_launchable_via_current_protocol": false, "verification_status": "static_only"}, + {"rank": 12, "username": "omegafantasy", "display_name": "刘家宏", "entity": "星之梦", "language": "make", "version": 27, "type": "cpp_source", "entry": "makefile -> main", "archive_sha256": "ea7be1c136d39404e9e788d609a16f0563762acb1cb542645049d05d1352dc67", "windows_native_runnable": "not_without_compile_and_mainexe_fix", "needs_compile": true, "needs_wsl_or_linux": false, "compiled_artifacts_present": false, "missing_files": false, "runner_launchable_via_current_protocol": false, "verification_status": "static_only", "skill_note": "接近但未解决"}, + {"rank": 13, "username": "ZarkLngeW", "display_name": "张龙文", "entity": "啥都没改", "language": "make", "version": 1, "type": "cpp_source", "entry": "makefile -> main", "archive_sha256": "3f0e72b410e1d065414327e1018a973aa98992240442d41466b0adea41c1a976", "windows_native_runnable": "not_without_compile_and_mainexe_fix", "needs_compile": true, "needs_wsl_or_linux": false, "compiled_artifacts_present": false, "missing_files": false, "runner_launchable_via_current_protocol": false, "verification_status": "static_only", "skill_note": "Archer lattice 研究断点之一"}, + {"rank": 14, "username": "ccz181078", "display_name": "蔡承泽", "entity": "sample", "language": "make", "version": 21, "type": "cpp_source", "entry": "makefile -> main", "archive_sha256": "1b1abdc0c8e619f20688627b6b49d41862dbac72c6ffd1a0671a9acb19abf256", "windows_native_runnable": "not_without_compile_and_mainexe_fix", "needs_compile": true, "needs_wsl_or_linux": false, "compiled_artifacts_present": false, "missing_files": false, "runner_launchable_via_current_protocol": false, "verification_status": "static_only"}, + {"rank": 15, "username": "tshoigyr", "display_name": "高宇睿", "entity": "a_idoit", "language": "make", "version": 21, "type": "cpp_source", "entry": "makefile -> main", "archive_sha256": "5170538ce62fbeacd7d837c2b2f3c6f8fe41a373c46ef1f497fa1bb2ac6773de", "windows_native_runnable": "not_without_compile_and_mainexe_fix", "needs_compile": true, "needs_wsl_or_linux": false, "compiled_artifacts_present": false, "missing_files": false, "runner_launchable_via_current_protocol": false, "verification_status": "static_only"}, + {"rank": 16, "username": "MoebiusMeow", "display_name": "郭佳琪", "entity": "起床大失败", "language": "make", "version": 1, "type": "cpp_source", "entry": "makefile -> main", "archive_sha256": "5f606babd6f99c8f1fe6b62ad959c233af01b44d6cc4e063510684db8785268a", "windows_native_runnable": "not_without_compile_and_mainexe_fix", "needs_compile": true, "needs_wsl_or_linux": false, "compiled_artifacts_present": false, "missing_files": false, "runner_launchable_via_current_protocol": false, "verification_status": "static_only", "skill_note": "Archer lattice 研究断点之一;由大败改善到接近平局但仍未获胜"} + ], + "_summary": { + "total": 16, + "python_script": ["rank04", "rank05", "rank07"], + "cpp_source": ["rank01", "rank02", "rank03", "rank06", "rank08", "rank09", "rank10", "rank11", "rank12", "rank13", "rank14", "rank15", "rank16"], + "windows_native_runnable_now": ["rank04", "rank05", "rank07"], + "needs_compile_plus_mainexe_fix_or_wsl": "13 (all cpp_source)", + "invalid_now": ["rank03"], + "runner_launchable_via_current_protocol": ["rank04", "rank05", "rank07"] + } +} diff --git a/src/agentbench_frame/games/miracle/matrix.py b/src/agentbench_frame/games/miracle/matrix.py new file mode 100644 index 0000000..851fb9f --- /dev/null +++ b/src/agentbench_frame/games/miracle/matrix.py @@ -0,0 +1,316 @@ +"""32-game matrix orchestrator logic for 24_miracle (Plan A). + +Pure, unit-tested logic (no Judge / no real match in this module): the fixed +32-attempt plan, atomic+resumable progress, per-game classification wiring, +per-rank audit gate, infrastructure-stop decision, independent PID+create_time +residual check, and final aggregation. The runner that drives real games lives +in ``tools/miracle_matrix.py`` and reuses these functions. + +Plan A (frozen): if-else Agent vs rank01–16, each opponent camp0 + camp1 once, +total attempts = 32. game_id = ``m_rank{NN}_camp{C}``; camp = the camp the +if-else Agent plays. +""" +from __future__ import annotations + +import json +import os +import re +from pathlib import Path +from typing import Any, Dict, List, Optional, Tuple + +from agentbench_frame.games.miracle.result import build_seed_provenance +from agentbench_frame.games.miracle.smoke_audit import ( + check_residual_procs, load_managed_procs_from_result_json, make_session_id, +) + +#: error_types that STOP the whole matrix (infrastructure failures). +#: ai_crash / ai_timeout are recorded invalid but do NOT stop (continue). +INFRA_STOP_ERRORS = { + "wrapper_timeout", "result_json_missing", "result_json_corrupt", + "evidence_mismatch", "replay_missing", "replay_corrupt", "judge_crash", + "cleanup_failure", "vendor_exception", +} + +_GAMEID_RE = re.compile(r"^m_rank(\d{2})_camp([01])$") + + +# --------------------------------------------------------------------------- # +# plan +# --------------------------------------------------------------------------- # +def make_attempt_plan() -> List[Dict[str, Any]]: + """32 attempts: rank01 camp0, rank01 camp1, ..., rank16 camp1.""" + plan = [] + for rank in range(1, 17): + for camp in (0, 1): + plan.append({"rank": rank, "camp": camp, + "game_id": f"m_rank{rank:02d}_camp{camp}"}) + return plan + + +def parse_game_id(game_id: str) -> Optional[Tuple[int, int]]: + m = _GAMEID_RE.match(game_id or "") + if not m: + return None + return int(m.group(1)), int(m.group(2)) + + +# --------------------------------------------------------------------------- # +# progress (atomic + resumable) +# --------------------------------------------------------------------------- # +def load_progress(path) -> Dict[str, Any]: + p = Path(path) + if not p.exists(): + return {"attempts": {}} + try: + return json.loads(p.read_text(encoding="utf-8")) + except (json.JSONDecodeError, OSError): + return {"attempts": {}} + + +def write_progress_atomic(path, progress: Dict[str, Any]) -> Path: + from agentbench_frame.games.miracle.atomicio import atomic_write_json + return atomic_write_json(path, progress) + + +def state_of(progress: Dict[str, Any], game_id: str) -> str: + return progress.get("attempts", {}).get(game_id, {}).get("state", "not_started") + + +def mark_running(progress: Dict[str, Any], game_id: str) -> None: + progress.setdefault("attempts", {})[game_id] = {"state": "running", "game_id": game_id} + + +def mark_done(progress: Dict[str, Any], game_id: str, result: Dict[str, Any]) -> None: + entry = {"state": "done", "game_id": game_id} + entry.update(result) + progress.setdefault("attempts", {})[game_id] = entry + + +def is_done(progress: Dict[str, Any], game_id: str) -> bool: + return state_of(progress, game_id) == "done" + + +def next_incomplete(progress: Dict[str, Any], plan: List[Dict[str, Any]]) -> Optional[Dict[str, Any]]: + for a in plan: + if not is_done(progress, a["game_id"]): + return a + return None + + +# --------------------------------------------------------------------------- # +# events append (no duplicate game_id) +# --------------------------------------------------------------------------- # +def append_event_atomic(path, event: Dict[str, Any]) -> bool: + """Append a JSON line; skip if its game_id already present (no rerun/dup).""" + p = Path(path) + gid = event.get("game_id") + existing = set() + if p.exists(): + for line in p.read_text(encoding="utf-8").splitlines(): + if line.strip(): + try: + e = json.loads(line) + if "game_id" in e: + existing.add(e["game_id"]) + except json.JSONDecodeError: + pass + if gid in existing: + return False + p.parent.mkdir(parents=True, exist_ok=True) + with p.open("a", encoding="utf-8") as f: + f.write(json.dumps(event, ensure_ascii=False) + "\n") + return True + + +# --------------------------------------------------------------------------- # +# per-game classification +# --------------------------------------------------------------------------- # +def classify_game(att) -> Dict[str, Any]: + """Map a MatchAttempt (with rank/camp or parseable game_id) to a per-game record.""" + rc = getattr(att, "rank", None), getattr(att, "camp", None) + if rc[0] is None or rc[1] is None: + parsed = parse_game_id(getattr(att, "game_id", "")) + rank = rc[0] if rc[0] is not None else (parsed[0] if parsed else None) + camp = rc[1] if rc[1] is not None else (parsed[1] if parsed else None) + else: + rank, camp = rc + rr = getattr(att, "realized_randomization", None) or {} + scores = getattr(att, "scores", None) or {} + return { + "game_id": att.game_id, + "rank": rank, + "camp": camp, + "ifelse_camp": camp, + "valid": bool(att.valid), + "normalized_result": att.normalized_result, + "raw_winner": getattr(att, "raw_winner", None), + "winner_agent": getattr(att, "winner_agent", None), + "scores": {"0": scores.get("0"), "1": scores.get("1")} if isinstance(scores, dict) else None, + "steps": int(getattr(att, "steps", 0) or 0), + "realized_randomization": rr, + "seed": build_seed_provenance(rr if rr else None), + "error_type": getattr(att, "error_type", None), + "reason": getattr(att, "reason", None), + "wrapper_timeout": bool(getattr(att, "wrapper_timeout", False)), + "ai_crash_player": getattr(att, "ai_crash_player", None), + "judge_crash": bool(getattr(att, "judge_crash", False)), + "normal_cleanup_nonzero": bool(getattr(att, "normal_cleanup_nonzero", False)), + "result_json_status": getattr(att, "result_json_status", "ok"), + "evidence_paths": getattr(att, "evidence_paths", {}), + } + + +# --------------------------------------------------------------------------- # +# stop decision +# --------------------------------------------------------------------------- # +def should_stop(att) -> Tuple[str, Optional[str]]: + """Infrastructure failures stop the matrix; AI crash/timeout (invalid) do NOT.""" + if getattr(att, "wrapper_timeout", False): + return ("stop", "wrapper_timeout") + et = getattr(att, "error_type", None) + if et in INFRA_STOP_ERRORS: + return ("stop", et) + rjs = getattr(att, "result_json_status", "ok") + if rjs in ("missing", "corrupt"): + return ("stop", f"result_json_{rjs}") + if getattr(att, "discrepancies", None): + return ("stop", "evidence_mismatch") + return ("continue", None) + + +def should_stop_with_residual(att) -> Tuple[str, str]: + """Like should_stop, plus an INDEPENDENT PID+create_time residual check on + the result-json's managed processes.""" + decision, reason = should_stop(att) + if decision == "stop": + return ("stop", reason) + rj_path = getattr(att, "evidence_paths", {}).get("result_json") + procs = load_managed_procs_from_result_json(rj_path) if rj_path else [] + if not procs: + return ("continue", "no_managed_procs_identity") + res = check_residual_procs(procs) + if res["residual"]: + return ("stop", f"residual_pids_{[(p.pid, p.role) for p in res['residual']]}") + return ("continue", "clean") + + +# --------------------------------------------------------------------------- # +# per-rank audit +# --------------------------------------------------------------------------- # +def rank_audit(rank: int, games) -> Tuple[bool, List[str]]: + reasons: List[str] = [] + recs = [g if isinstance(g, dict) else classify_game(g) for g in games] + if len(recs) != 2: + reasons.append(f"rank{rank}: expected 2 games, got {len(recs)}") + camps = sorted(r.get("camp") for r in recs if r.get("camp") is not None) + if camps != [0, 1]: + reasons.append(f"rank{rank}: camps={camps} != [0,1]") + ids = [r.get("game_id") for r in recs] + if len(set(ids)) != len(ids): + reasons.append(f"rank{rank}: duplicate game_id {ids}") + for r in recs: + if r.get("valid") and r.get("normalized_result") not in ("win", "loss", "draw"): + reasons.append(f"{r.get('game_id')}: valid but normalized_result={r.get('normalized_result')}") + if not r.get("valid") and not r.get("error_type"): + reasons.append(f"{r.get('game_id')}: invalid without error_type") + return (len(reasons) == 0, reasons) + + +# --------------------------------------------------------------------------- # +# aggregation +# --------------------------------------------------------------------------- # +def _safe_num(x): + try: + return float(x) + except (TypeError, ValueError): + return None + + +def aggregate(records: List[Dict[str, Any]]) -> Dict[str, Any]: + """Final matrix aggregate. win_rate = valid_wins / valid_games (null if 0).""" + valid = [r for r in records if r.get("valid")] + wins = [r for r in valid if r.get("normalized_result") == "win"] + losses = [r for r in valid if r.get("normalized_result") == "loss"] + invalid = [r for r in records if not r.get("valid")] + valid_games = len(valid) + + # per-rank + per_rank: Dict[int, Dict[str, Any]] = {} + for rank in range(1, 17): + rs = [r for r in records if r.get("rank") == rank] + rv = [r for r in rs if r.get("valid")] + rw = [r for r in rv if r.get("normalized_result") == "win"] + rl = [r for r in rv if r.get("normalized_result") == "loss"] + per_rank[rank] = { + "attempts": len(rs), "valid": len(rv), "invalid": len(rs) - len(rv), + "wins": len(rw), "losses": len(rl), + "win_rate": (len(rw) / len(rv)) if rv else None, + } + + # camp split + camp_split = {} + for camp in (0, 1): + cv = [r for r in valid if r.get("camp") == camp] + cw = [r for r in cv if r.get("normalized_result") == "win"] + camp_split[camp] = {"valid": len(cv), "wins": len(cw), + "win_rate": (len(cw) / len(cv)) if cv else None} + + # score stats (valid games, ifelse score minus opponent score) + diffs = [] + for r in valid: + s = r.get("scores") or {} + s0, s1 = _safe_num(s.get("0")), _safe_num(s.get("1")) + if s0 is not None and s1 is not None: + camp = r.get("camp") + diffs.append(s0 - s1 if camp == 0 else s1 - s0) # from ifelse perspective + diffs_sorted = sorted(diffs) + def median(xs): + n = len(xs) + if n == 0: + return None + return xs[n // 2] if n % 2 else (xs[n // 2 - 1] + xs[n // 2]) / 2 + score_stats = { + "n": len(diffs), + "mean_diff": (sum(diffs) / len(diffs)) if diffs else None, + "median_diff": median(diffs_sorted), + } + + # steps stats (valid) + vsteps = [int(r.get("steps", 0) or 0) for r in valid] + steps_stats = { + "n": len(vsteps), + "total": sum(vsteps), + "mean": (sum(vsteps) / len(vsteps)) if vsteps else None, + "median": median(sorted(vsteps)), + } + + # realized randomization distribution (valid games) + map_dist = {"map_type": {}, "day_time": {}} + for r in valid: + rr = r.get("realized_randomization") or {} + for k in ("map_type", "day_time"): + v = rr.get(k) + if v is not None: + map_dist[k][v] = map_dist[k].get(v, 0) + 1 + + # invalid reason distribution + invalid_reasons: Dict[str, int] = {} + for r in invalid: + et = r.get("error_type") or "unknown" + invalid_reasons[et] = invalid_reasons.get(et, 0) + 1 + + return { + "total_attempts": len(records), + "valid_games": valid_games, + "invalid_games": len(invalid), + "wins": len(wins), + "losses": len(losses), + "win_rate": (len(wins) / valid_games) if valid_games else None, + "per_rank": per_rank, + "camp_split": camp_split, + "score_stats": score_stats, + "steps_stats": steps_stats, + "realized_randomization_distribution": map_dist, + "invalid_reason_distribution": invalid_reasons, + "small_sample_note": "32 局为小样本;含 invalid;不把对手崩溃包装为 Agent 胜利;win_rate 分母仅 valid games。", + } diff --git a/src/agentbench_frame/games/miracle/matrix_runner.py b/src/agentbench_frame/games/miracle/matrix_runner.py new file mode 100644 index 0000000..2b41a90 --- /dev/null +++ b/src/agentbench_frame/games/miracle/matrix_runner.py @@ -0,0 +1,945 @@ +"""Matrix runner: drives the 32-game Plan A using the tested matrix.py logic. + +Designed for ONE long-running process (single background runner). Safety: + * ``--dry-run`` / ``dry_run()`` never calls attempt_fn (no Judge/AI). + * Unique session; existing session refused (no overwrite/delete). + * Atomic manifest written before game 1. + * ``running`` entries are UNCERTAIN_IN_FLIGHT on restart → STOP, never auto-rerun. + * ``done`` written only after the attempt returned + event landed. + * valid / AI-invalid (continue, not in win-rate denom) / infra-failure (halt) distinct. + * Per-rank audit after both camps; PID+create_time residual halts. + * events.jsonl is the source of truth → summary independently re-computable. + +The real attempt_fn wraps ``match_runner.run_match_attempt``; tests inject a fake. +""" +from __future__ import annotations + +import json +import os +from pathlib import Path +from typing import Any, Callable, Dict, List, Optional + +from agentbench_frame.games.miracle import matrix +from agentbench_frame.games.miracle.matrix import ( + aggregate, + append_event_atomic, + classify_game, + is_done, + load_progress, + make_attempt_plan, + make_session_id, + mark_done, + mark_running, + parse_game_id, + rank_audit, + should_stop_with_residual, + write_progress_atomic, +) +from agentbench_frame.games.miracle.smoke_audit import ensure_fresh_session + +# re-export for tests +__all__ = ["MatrixRunner", "make_attempt_plan", "parse_game_id", "mark_running", + "write_progress_atomic", "has_uncertain"] + + +def has_uncertain(progress: Dict[str, Any]) -> List[str]: + """game_ids left in 'running' state — uncertain whether they played.""" + return [gid for gid, e in progress.get("attempts", {}).items() if e.get("state") == "running"] + + +# --------------------------------------------------------------------------- # +# strict manifest / progress loaders (review#2 严格性收口) +# --------------------------------------------------------------------------- # +# The 32 standard Plan A game_ids. plan_in_manifest must overwhelmingly equal +# this set; used to reject unknown progress entries when no expected_plan is +# available and to derive the known-gid set. +_LEGAL_STATES = ("not_started", "running", "done") + + +def _known_gids_from_plan(*plans) -> set: + """Derive the authoritative set of game_ids from one or more plan lists.""" + out: set = set() + for p in plans: + if p: + for a in p: + if isinstance(a, dict) and a.get("game_id") is not None: + out.add(a["game_id"]) + return out + + +def _load_progress_strict(path) -> Dict[str, Any]: + """Strict progress.json parse — NEVER swallow errors. Raises if the file is + present but corrupt/non-dict so resume() cannot accidentally proceed on a + half-broken progress that the prior verify_session_for_resume already + flagged (review#2 gap 9: no lenient re-parse after strict verify).""" + p = Path(path) + if not p.exists(): + return {"attempts": {}} + try: + data = json.loads(p.read_text(encoding="utf-8")) + except (json.JSONDecodeError, ValueError, UnicodeDecodeError) as e: + raise RuntimeError(f"progress.json corrupt/unparseable: {e}") from e + if not isinstance(data, dict): + raise RuntimeError(f"progress.json not a JSON object: got {type(data).__name__}") + if not isinstance(data.get("attempts", {}), dict): + raise RuntimeError("progress.attempts is not a dict") + return data + + +def _load_manifest_strict(path) -> Dict[str, Any]: + """Strict manifest.json parse for resume(): raises on corrupt/non-dict so + a broken manifest cannot silently let a wrong run_id through.""" + p = Path(path) + if not p.exists(): + return {} + try: + data = json.loads(p.read_text(encoding="utf-8")) + except (json.JSONDecodeError, ValueError, UnicodeDecodeError) as e: + raise RuntimeError(f"manifest.json corrupt/unparseable: {e}") from e + if not isinstance(data, dict): + raise RuntimeError(f"manifest.json not a JSON object: got {type(data).__name__}") + return data + + +def verify_session_for_resume(session_dir, *, + protocol_sha: Optional[str] = None, + code_files: Optional[Dict[str, str]] = None, + expected_plan: Optional[List[Dict]] = None, + expected_ifelse_sha: Optional[str] = None, + expected_judge_sha: Optional[str] = None, + expected_opponent_shas: Optional[Dict[int, str]] = None, + expected_build_shas: Optional[Dict[int, str]] = None, + expected_python: Optional[str] = None, + expected_platform: Optional[str] = None, + expected_control_inputs: Optional[Dict[str, str]] = None) -> Tuple[bool, List[str]]: + """Full READ-ONLY session verification before resume. Returns (ok, errors). + Does NOT write or modify anything. If ANY check fails, the session must be + left byte-for-byte unchanged.""" + import hashlib + sd = Path(session_dir) + errs: List[str] = [] + + # --- manifest --- + mp = sd / "manifest.json" + if not mp.exists(): + return (False, ["manifest missing"]) + try: + m = json.loads(mp.read_text(encoding="utf-8")) + except Exception: + return (False, ["manifest corrupt/unparseable"]) + if expected_control_inputs is not None: + stored_inputs = m.get("control_inputs") + if not isinstance(stored_inputs, dict): + errs.append("control input hash mismatch: control_inputs missing") + else: + expected_names = set(expected_control_inputs) + stored_names = set(stored_inputs) + for name in sorted(expected_names - stored_names): + errs.append(f"control input hash mismatch: {name} missing") + for name in sorted(stored_names - expected_names): + errs.append(f"control input hash mismatch: {name} unexpected") + for name in sorted(expected_names & stored_names): + item = stored_inputs[name] + stored_sha = item.get("sha256") if isinstance(item, dict) else None + if stored_sha != expected_control_inputs[name]: + errs.append(f"control input hash mismatch: {name}") + if protocol_sha and m.get("protocol_sha256") != protocol_sha: + errs.append(f"protocol sha mismatch") + if code_files: + stored = m.get("code_hashes", {}) + for name, p in code_files.items(): + p = Path(p) + if not p.exists(): + errs.append(f"code file missing: {name}") + elif stored.get(name) != hashlib.sha256(p.read_bytes()).hexdigest(): + errs.append(f"code hash mismatch: {name}") + if m.get("timeout") != 8.0: + errs.append(f"timeout mismatch: {m.get('timeout')}") + _wts = m.get("wrapper_timeout_s") + if _wts is None: + errs.append("wrapper_timeout_s missing or null") + elif _wts != 180.0: + errs.append(f"wrapper_timeout_s mismatch: {_wts}") + # --- asset hashes (identity frozen at session creation) --- + if expected_ifelse_sha and m.get("ifelse_sha256") != expected_ifelse_sha: + errs.append("if-else sha mismatch") + if expected_judge_sha and m.get("judge_sha256") != expected_judge_sha: + errs.append("judge sha mismatch") + if expected_opponent_shas: + stored_opp = m.get("opponent_archive_sha256", {}) + for rk, v in expected_opponent_shas.items(): + key = f"rank{rk:02d}" if isinstance(rk, int) else str(rk) + if stored_opp.get(key) != v: + errs.append(f"opponent archive sha mismatch: {key}") + if expected_build_shas: + stored_build = m.get("cpp_build_sha256", {}) + for rk, v in expected_build_shas.items(): + key = f"rank{rk:02d}" if isinstance(rk, int) else str(rk) + if stored_build.get(key) != v: + errs.append(f"build sha mismatch: {key}") + # --- plan_count must equal len(plan) AND (when len-plan fallback) 32 --- + plan_in_manifest = m.get("plan", []) + if not isinstance(plan_in_manifest, list): + errs.append("plan field is not a list") + plan_in_manifest = [] + plan_len = len(plan_in_manifest) + declared_count = m.get("plan_count") + if declared_count is None: + errs.append("plan_count field missing") + elif not isinstance(declared_count, int) or declared_count != plan_len: + errs.append(f"plan_count={declared_count!r} != len(plan)={plan_len}") + # --- plan: full 32 content + order --- + if expected_plan: + if plan_len != len(expected_plan): + errs.append(f"plan length {plan_len} != {len(expected_plan)}") + else: + for i, (got, want) in enumerate(zip(plan_in_manifest, expected_plan)): + if not isinstance(got, dict): + errs.append(f"plan[{i}] not a dict: {got!r}") + continue + if got.get("game_id") != want.get("game_id"): + errs.append(f"plan[{i}] game_id {got.get('game_id')} != {want.get('game_id')}") + if got.get("rank") != want.get("rank"): + errs.append(f"plan[{i}] rank {got.get('rank')} != {want.get('rank')}") + if got.get("camp") != want.get("camp"): + errs.append(f"plan[{i}] camp {got.get('camp')} != {want.get('camp')}") + # reject extra/unexpected keys in the per-attempt plan entry + _want_keys = set(want.keys()) + _got_keys = set(got.keys()) + _extra = _got_keys - _want_keys + if _extra: + errs.append(f"plan[{i}] unexpected keys: {sorted(_extra)}") + elif plan_len != 32: + errs.append(f"plan_count={plan_len} != 32 (len(plan) fallback)") + # --- session_id MUST exist (review#2 gap 4: missing → reject) --- + m_sid = m.get("session_id") + if not m_sid: + errs.append("session_id missing or empty") + elif m_sid != sd.name: + errs.append(f"session_id mismatch: manifest={m_sid} != dir={sd.name}") + # --- run_id MUST exist --- + if not m.get("run_id"): + errs.append("run_id missing") + # --- runtime identity --- + if expected_python and m.get("python_version") and m.get("python_version") != expected_python: + errs.append(f"python version mismatch: {m.get('python_version')} != {expected_python}") + if expected_platform and m.get("platform") and m.get("platform") != expected_platform: + errs.append(f"platform mismatch: {m.get('platform')} != {expected_platform}") + # --- opponent / build / code_hashes: EXACT key-set match (review#2 gap 3) --- + if expected_opponent_shas is not None: + stored_opp = m.get("opponent_archive_sha256", {}) + if not isinstance(stored_opp, dict): + errs.append("opponent_archive_sha256 not a dict") + else: + expected_keys = {f"rank{rk:02d}" if isinstance(rk, int) else str(rk) for rk in expected_opponent_shas} + extra = set(stored_opp.keys()) - expected_keys + missing = expected_keys - set(stored_opp.keys()) + if extra: + errs.append(f"opponent_archive_sha256 unexpected/extra keys: {sorted(extra)}") + if missing: + errs.append(f"opponent_archive_sha256 missing keys: {sorted(missing)}") + if expected_build_shas is not None: + stored_build = m.get("cpp_build_sha256", {}) + if not isinstance(stored_build, dict): + errs.append("cpp_build_sha256 not a dict") + else: + expected_keys = {f"rank{rk:02d}" if isinstance(rk, int) else str(rk) for rk in expected_build_shas} + extra = set(stored_build.keys()) - expected_keys + missing = expected_keys - set(stored_build.keys()) + if extra: + errs.append(f"cpp_build_sha256 unexpected/extra keys: {sorted(extra)}") + if missing: + errs.append(f"cpp_build_sha256 missing keys: {sorted(missing)}") + if code_files is not None: + stored_code = m.get("code_hashes", {}) + if not isinstance(stored_code, dict): + errs.append("code_hashes not a dict") + else: + expected_keys = set(code_files.keys()) + extra = set(stored_code.keys()) - expected_keys + missing = expected_keys - set(stored_code.keys()) + if extra: + errs.append(f"code_hashes unexpected/extra keys: {sorted(extra)}") + if missing: + errs.append(f"code_hashes missing keys: {sorted(missing)}") + + # --- progress / events / audit consistency --- + pp = sd / "progress.json" + ep = sd / "events.jsonl" + ad = sd / "audit" + # STRICT progress parsing (do NOT use lenient load_progress which swallows errors) + if not pp.exists(): + errs.append("progress.json missing") + attempts = {} + else: + try: + progress = json.loads(pp.read_text(encoding="utf-8")) + except Exception: + errs.append("progress.json corrupt") + progress = {} + if not isinstance(progress, dict): + errs.append("progress.json not a dict") + progress = {} + attempts = progress.get("attempts", {}) + if not isinstance(attempts, dict): + errs.append("progress attempts not a dict") + attempts = {} + # events.jsonl — STRICT line-by-line: each non-empty line must be a JSON OBJECT + event_gids: List[str] = [] + if ep.exists(): + for line_no, line in enumerate(ep.read_text(encoding="utf-8").splitlines(), 1): + if not line.strip(): + continue + try: + e = json.loads(line) + except json.JSONDecodeError: + errs.append(f"events.jsonl line {line_no}: corrupt (not JSON)") + continue + if not isinstance(e, dict): + errs.append(f"events.jsonl line {line_no}: top-level not a JSON object") + continue + if "game_id" in e: + event_gids.append(e["game_id"]) + # --- progress attempts: known gids + legal state (review#2 gaps 1, 2) --- + known_gids = _known_gids_from_plan(expected_plan, plan_in_manifest) + if not known_gids: + # expected_plan not provided AND plan was rejected; use the std 32. + known_gids = {a["game_id"] for a in make_attempt_plan()} + for gid, entry in attempts.items(): + if not isinstance(entry, dict): + errs.append(f"progress attempt {gid!r}: entry not a dict") + continue + if gid not in known_gids: + errs.append(f"progress entry unknown game_id: {gid}") + continue + state = entry.get("state", "not_started") + if state not in _LEGAL_STATES: + errs.append(f"progress {gid}: illegal state {state!r}") + # running → UNCERTAIN + running = [gid for gid, e in attempts.items() if isinstance(e, dict) and e.get("state") == "running"] + if running: + errs.append(f"UNCERTAIN_IN_FLIGHT: {running}") + # done → exactly one event + done_gids = {gid for gid, e in attempts.items() if isinstance(e, dict) and e.get("state") == "done"} + from collections import Counter + ev_counts = Counter(event_gids) + dups = {gid: c for gid, c in ev_counts.items() if c > 1} + if dups: + errs.append(f"duplicate events: {dups}") + for gid in done_gids: + if ev_counts.get(gid, 0) != 1: + errs.append(f"done {gid} has {ev_counts.get(gid, 0)} events (expected 1)") + # not_started must not have events + not_started_with_ev = [gid for gid in ev_counts if gid not in done_gids and gid not in running] + if not_started_with_ev: + errs.append(f"not_started with events: {not_started_with_ev}") + # --- per-rank audit (review#2 gaps 6, 7) --- + for rank in range(1, 17): + c0 = f"m_rank{rank:02d}_camp0" + c1 = f"m_rank{rank:02d}_camp1" + both_done = c0 in done_gids and c1 in done_gids + af = ad / f"rank{rank:02d}.json" + if af.exists() and not both_done: + # premature/partial audit: audit present while neither OR only one camp done + only_one = (c0 in done_gids) ^ (c1 in done_gids) + errs.append(f"rank{rank:02d} premature/partial complete audit present " + f"(both_done={both_done} only_one_done={only_one})") + continue + if not af.exists(): + if both_done: + errs.append(f"complete rank{rank:02d} missing audit file") + continue + # audit present AND both done → deep validation (review#2 gap 6) + try: + audit = json.loads(af.read_text(encoding="utf-8")) + except Exception: + errs.append(f"rank{rank:02d} audit corrupt/unparseable") + continue + if not isinstance(audit, dict): + errs.append(f"rank{rank:02d} audit not a JSON object") + continue + if audit.get("rank") != rank: + errs.append(f"rank{rank:02d} audit rank mismatch: {audit.get('rank')!r}") + if audit.get("ok") is not True: + errs.append(f"rank{rank:02d} audit ok is not True (got {audit.get('ok')!r})") + audit_games = audit.get("games") + audit_pairs: List[Tuple] = [] + if isinstance(audit_games, list): + for g in audit_games: + if isinstance(g, dict): + audit_pairs.append((g.get("game_id"), g.get("camp"))) + expected_pairs = {(c0, 0), (c1, 1)} + actual_set = set(audit_pairs) + if actual_set != expected_pairs: + errs.append(f"rank{rank:02d} audit game_id/camp mismatch: " + f"{sorted(map(str, actual_set))} != {sorted(map(str, expected_pairs))}") + + return (len(errs) == 0, errs) + + +class MatrixRunner: + def __init__(self, *, session_root, judge_dir, ifelse_dir, opponent_dir_of, + vendor_script, framework_src, timeout: float = 8.0, + wrapper_timeout_s: float = 60.0, attempt_fn: Optional[Callable] = None, + protocol_sha: str = "", run_id: Optional[str] = None, + python: Optional[str] = None, evaluated_agent: str = "miracle_ifelse", + auth_text: str = ""): + self.session_root = Path(session_root) + self.judge_dir = Path(judge_dir) + self.ifelse_dir = Path(ifelse_dir) + self.opponent_dir_of = opponent_dir_of + self.vendor_script = Path(vendor_script) + self.framework_src = Path(framework_src) + self.timeout = timeout + self.wrapper_timeout_s = wrapper_timeout_s + self.attempt_fn = attempt_fn or self._default_attempt_fn + self.protocol_sha = protocol_sha + self.run_id = run_id or make_session_id() + self.python = python + self.evaluated_agent = evaluated_agent + self.auth_text = auth_text + self.session_id: Optional[str] = None + self.session_dir: Optional[Path] = None + self.progress: Dict[str, Any] = {"attempts": {}} + self.plan = make_attempt_plan() + + # ---- session / paths ---- # + def _setup_paths(self): + self.work_dir = self.session_dir / "work" + self.events_path = self.session_dir / "events.jsonl" + self.progress_path = self.session_dir / "progress.json" + self.data_dir = self.session_dir / "data" + self.run_dir = self.data_dir / "runs" / "24_miracle" / self.evaluated_agent / self.run_id + self.work_dir.mkdir(parents=True, exist_ok=True) + self.data_dir.mkdir(parents=True, exist_ok=True) + + def prepare_session(self) -> Path: + self.session_id = make_session_id() + self.session_dir = ensure_fresh_session(self.session_root, self.session_id) + self._setup_paths() + write_progress_atomic(self.progress_path, self.progress) + return self.session_dir + + def prepare_session_for_existing(self, session_id: str) -> Path: + sd = self.session_root / session_id + if sd.exists(): + raise FileExistsError(f"session already exists; refusing to overwrite: {sd}") + self.session_id = session_id + self.session_dir = ensure_fresh_session(self.session_root, session_id) + self._setup_paths() + return self.session_dir + + def resume(self, session_id: str) -> Path: + """Open an EXISTING session for resumption (no new session created). + Restores the original run_id from manifest BEFORE _setup_paths so the + run_dir matches the original session, not a new run_id. + + Uses STRICT manifest+progress parse (review#2 gap 9): a corrupt file + must RAISE, never silently swallow on lenient re-parse after a strict + verify_session_for_resume already approved the session. + """ + sd = self.session_root / session_id + if not sd.exists(): + raise FileNotFoundError(f"cannot resume: session not found: {sd}") + # restore run_id from manifest BEFORE _setup_paths (which builds run_dir from run_id) + _m = _load_manifest_strict(sd / "manifest.json") + _rid = _m.get("run_id") if _m else None + if _rid: + self.run_id = _rid + self.session_id = session_id + self.session_dir = sd + self._setup_paths() + # STRICT progress parse — never swallow errors after a successful verify + self.progress = _load_progress_strict(self.progress_path) + return self.session_dir + + # ---- manifest ---- # + def record_manifest(self, *, opponent_hashes: Dict[int, str], build_hashes: Dict[int, str], + ifelse_sha: str, judge_sha: str, code_hashes: Dict[str, str], + control_inputs: Optional[Dict[str, Dict[str, str]]] = None, + ) -> Path: + import platform, time + m = { + "session_id": self.session_id, "run_id": self.run_id, + "created_unix": time.time(), + "protocol_sha256": self.protocol_sha, "auth_text": self.auth_text, + "plan_count": len(self.plan), "plan": self.plan, + "timeout": self.timeout, "wrapper_timeout_s": self.wrapper_timeout_s, + "evaluated_agent": self.evaluated_agent, + "ifelse_sha256": ifelse_sha, "judge_sha256": judge_sha, + "opponent_archive_sha256": {f"rank{k:02d}": v for k, v in opponent_hashes.items()}, + "cpp_build_sha256": {f"rank{k:02d}": v for k, v in build_hashes.items()}, + "code_hashes": code_hashes, + "platform": platform.platform(), + } + if control_inputs is not None: + m["control_inputs"] = control_inputs + mp = self.session_dir / "manifest.json" + tmp = mp.with_suffix(".json.tmp") + tmp.write_text(json.dumps(m, ensure_ascii=False, indent=2), encoding="utf-8") + os.replace(tmp, mp) + return mp + + # ---- dry run (no subprocess) ---- # + def dry_run(self) -> Dict[str, Any]: + return {"session_id": self.session_id, "run_id": self.run_id, + "plan_count": len(self.plan), "plan": self.plan, + "judge_dir": str(self.judge_dir), "ifelse_dir": str(self.ifelse_dir), + "evaluated_agent": self.evaluated_agent, "timeout": self.timeout, + "wrapper_timeout_s": self.wrapper_timeout_s} + + # ---- per-game ---- # + def _attempt_kwargs(self, attempt: Dict[str, Any]) -> Dict[str, Any]: + rank, camp = attempt["rank"], attempt["camp"] + opp_dir = self.opponent_dir_of(rank) + opp_name = f"rank{rank:02d}" + if camp == 0: + p0_dir, p1_dir = self.ifelse_dir, opp_dir + p0_name, p1_name = self.evaluated_agent, opp_name + else: + p0_dir, p1_dir = opp_dir, self.ifelse_dir + p0_name, p1_name = opp_name, self.evaluated_agent + return dict(game_id=attempt["game_id"], p0_dir=p0_dir, p1_dir=p1_dir, + p0_name=p0_name, p1_name=p1_name, evaluated_agent_camp=camp, + evaluated_agent=self.evaluated_agent, opponent=opp_name, + work_dir=self.work_dir, judge_dir=self.judge_dir, + vendor_script=self.vendor_script, framework_src=self.framework_src, + timeout=self.timeout, wrapper_timeout_s=self.wrapper_timeout_s, + python=self.python) + + def _default_attempt_fn(self, **kw): + from agentbench_frame.games.miracle.match_runner import run_match_attempt + return run_match_attempt(**kw) + + def _run_one(self, attempt: Dict[str, Any]) -> Dict[str, Any]: + gid = attempt["game_id"] + self.progress = load_progress(self.progress_path) # resumable read + if is_done(self.progress, gid): + return {"halted": False, "skipped": True, "game_id": gid} + uncertain = has_uncertain(self.progress) + if uncertain: + return {"halted": True, "state": "HALTED_INFRA_FAILURE", + "reason": f"UNCERTAIN_IN_FLIGHT: {uncertain}"} + mark_running(self.progress, gid) + write_progress_atomic(self.progress_path, self.progress) + att = self.attempt_fn(**self._attempt_kwargs(attempt)) + rec = classify_game(att) + append_event_atomic(self.events_path, rec) + decision, reason = should_stop_with_residual(att) + if decision == "stop": + rec["halt_reason"] = reason + mark_done(self.progress, gid, rec) + write_progress_atomic(self.progress_path, self.progress) + return {"halted": True, "state": "HALTED_INFRA_FAILURE", + "reason": reason, "record": rec, "game_id": gid} + mark_done(self.progress, gid, rec) + write_progress_atomic(self.progress_path, self.progress) + return {"halted": False, "record": rec, "game_id": gid} + + # ---- per-rank ---- # + def execute_rank(self, rank: int) -> Dict[str, Any]: + attempts = [a for a in self.plan if a["rank"] == rank] + audit_dir = self.session_dir / "audit" + audit_dir.mkdir(parents=True, exist_ok=True) + for att in attempts: + res = self._run_one(att) + if res.get("halted"): + return res + # rank audit + self.progress = load_progress(self.progress_path) + games = [self.progress["attempts"][a["game_id"]] for a in attempts] + ok, reasons = rank_audit(rank, games) + (audit_dir / f"rank{rank:02d}.json").write_text( + json.dumps({"rank": rank, "ok": ok, "reasons": reasons, + "games": games}, ensure_ascii=False, indent=2), encoding="utf-8") + if not ok: + return {"halted": True, "state": "HALTED_INFRA_FAILURE", + "reason": f"rank{rank:02d} audit failed: {reasons}"} + return {"halted": False, "rank": rank} + + # ---- whole matrix ---- # + def execute(self) -> Dict[str, Any]: + self.progress = load_progress(self.progress_path) + uncertain = has_uncertain(self.progress) + if uncertain: + return {"halted": True, "state": "HALTED_INFRA_FAILURE", + "reason": f"UNCERTAIN_IN_FLIGHT: {uncertain}"} + for rank in range(1, 17): + rank_atts = [a for a in self.plan if a["rank"] == rank] + if all(is_done(self.progress, a["game_id"]) for a in rank_atts): + continue + res = self.execute_rank(rank) + if res.get("halted"): + return res + return {"halted": False, "completed": True, "state": "COMPLETE"} + + # ---- aggregate from events (independently re-computable) ---- # + def aggregate_from_events(self) -> Dict[str, Any]: + records = [] + if self.events_path.exists(): + for line in self.events_path.read_text(encoding="utf-8").splitlines(): + if line.strip(): + try: + records.append(json.loads(line)) + except json.JSONDecodeError: + pass + return aggregate(records) + + # ---- Run-compatible output for Results pipeline (review #1) ---- # + def write_run_compatible_output(self) -> Path: + """Drive the framework ``Run`` lifecycle to produce CI-compatible + run output. + + Review #1 §5 + §6 收口:使用 staging → validation → atomic promotion + 模式。**绝不破坏性覆写**既有 run 目录。 + + 步骤: + 1. 在 *同一文件系统* 的 staging 目录(``/.staging/``) + 下生成完整候选 run(events.jsonl + summary.json + run.toml)。 + 2. 对候选 run 完整校验:32 unique game_ids(计划数)、event quality + 合法、run_id 一致性、events↔summary 重算一致、totals 通过 + ``Run.recompute_totals_from_events()`` 来自真实事件计数。 + 3. 校验全部通过后,对既有 live run 若存在则备份到 + ``/.backup_``(同文件系统),然后原子 os.replace + 交换 staging ↔ live。 + 4. promotion 失败时从备份回滚,不留下半截 events。 + 5. 最终 live 目录正好含一个 run_id 目录(不生成第二个)。 + + framework envelope 由 ``Run.write("game", **rec)`` 自动填,Miracle + 一手字段逐字透传。 + """ + from agentbench_frame.tracking.run import Run + from agentbench_frame.tracking.quality import inspect_event_file + + agg = self.aggregate_from_events() + + # 1) staging directory: /.staging/ + # same fs as data_dir so os.replace is atomic. + staging_root = self.data_dir / ".staging" + staging_root.mkdir(parents=True, exist_ok=True) + staging = staging_root / self.run_id + live_run_dir = self.run_dir + # A previous writer may have been terminated between the two directory + # renames. Recover before deleting/reusing any staging path. + self._recover_promotion_transaction( + live_run_dir, + live_run_dir.with_name(live_run_dir.name + ".backup_promote"), + live_run_dir.with_name(live_run_dir.name + ".promote_marker.json"), + ) + # always start staging clean (this is a scratch path, not run storage) + if staging.exists(): + import shutil as _sh + _sh.rmtree(staging) + # build a Run inside staging (NOT in the final runs/// + # path): we point data_dir at staging so Run writes its files there. + run = Run.start(game="24_miracle", agent=self.evaluated_agent, + run_type="eval", data_dir=str(staging), + run_id=self.run_id, append=False, + config={"matrix": "plan_a_32", + "total_attempts": agg["total_attempts"]}) + + # 2) re-emit every matrix session event through framework envelope + if self.events_path.exists(): + for line in self.events_path.read_text(encoding="utf-8").splitlines(): + if not line.strip(): + continue + try: + rec = json.loads(line) + except json.JSONDecodeError: + continue + if not isinstance(rec, dict): + continue + run.write("game", **rec) + + # 3) recompute totals from valid game events on disk + recomputed = run.recompute_totals_from_events(game_event_type="game") + # log h2h (matrix synth-extracted) for summary wpis + run.log_h2h(agg.get("h2h", {}) if isinstance(agg, dict) else {}) + + # 4) finish in staging + run.finish(extra_summary={ + "win_rate": agg["win_rate"], + "win_rate_available": (agg["valid_games"] > 0), + "total_episodes": recomputed["episodes"], + "total_steps": recomputed["total_steps"], + "matrix_aggregate": agg, + "wins": recomputed["wins"], + "losses": recomputed["losses"], + "evaluation_status": + ("COMPLETE" if agg["valid_games"] > 0 else "NO_VALID_GAMES"), + }) + + # 5) validate candidate BEFORE promotion + candidate_run_dir = staging / "runs" / "24_miracle" / self.evaluated_agent / self.run_id + errs = self._validate_candidate_run(candidate_run_dir, expected_count=agg["total_attempts"]) + if errs: + # leave staging in place for inspection (it's under .staging/, not + # the real runs/ path), but do NOT promote. Caller-visible live run + # (if any) is byte-identical to its prior state. + raise RuntimeError(f"candidate run validation failed: {errs}") + + # 6) atomic promotion: staging → live, with backup-and-rollback. + self._atomic_promote(candidate_run_dir, live_run_dir) + self._cleanup_staging(staging_root, self.run_id) + return live_run_dir + + # ---- promotion helpers ------------------------------------------------ # + @staticmethod + def _validate_candidate_run(run_dir: Path, *, + expected_count: int) -> List[str]: + """Validate the candidate run directory prior to promotion. Returns + a list of failure descriptions (empty list = candidate valid).""" + errs: List[str] = [] + if not run_dir.exists(): + return [f"candidate run dir absent: {run_dir}"] + for fname in ("events.jsonl", "summary.json", "run.toml"): + if not (run_dir / fname).exists(): + errs.append(f"candidate missing: {fname}") + if errs: + return errs + # event quality + from agentbench_frame.tracking.quality import inspect_event_file + rep = inspect_event_file(run_dir / "events.jsonl") + if rep.malformed_lines: + errs.append(f"candidate event_quality: malformed_lines={rep.malformed_lines}") + if rep.duplicate_event_ids: + errs.append(f"candidate event_quality: duplicates={rep.duplicate_event_ids}") + if rep.missing_event_ids: + errs.append(f"candidate event_quality: missing event_ids={rep.missing_event_ids}") + if rep.missing_run_ids: + errs.append(f"candidate event_quality: missing run_ids={rep.missing_run_ids}") + # 32 (or expected_count) unique game_ids + ev = [] + for line in (run_dir / "events.jsonl").read_text(encoding="utf-8").splitlines(): + if line.strip(): + try: + ev.append(json.loads(line)) + except json.JSONDecodeError: + pass + game_events = [e for e in ev + if isinstance(e, dict) + and (e.get("event_type") == "game" or e.get("event") == "game")] + gids = [e.get("game_id") for e in game_events if e.get("game_id") is not None] + unique = set(gids) + if len(unique) != expected_count: + errs.append(f"candidate unique game_ids={len(unique)} != expected {expected_count}") + if len(gids) != len(unique): + errs.append(f"candidate duplicate game_ids in events.jsonl={len(gids)-len(unique)}") + # summary.json wins/losses/total_episodes == recomputed from events + s = json.loads((run_dir / "summary.json").read_text(encoding="utf-8")) + valid = [e for e in game_events if e.get("valid")] + wins_ev = sum(1 for e in valid if e.get("normalized_result") == "win") + loss_ev = sum(1 for e in valid if e.get("normalized_result") == "loss") + steps_ev = sum(int(e.get("steps") or 0) for e in valid) + if s.get("total_episodes") != len(valid): + errs.append(f"summary.total_episodes={s.get('total_episodes')} != valid_events={len(valid)}") + if s.get("total_steps") != steps_ev: + errs.append(f"summary.total_steps={s.get('total_steps')} != events_steps_sum={steps_ev}") + if s.get("wins") != wins_ev: + errs.append(f"summary.wins={s.get('wins')} != events_wins={wins_ev}") + if s.get("losses") != loss_ev: + errs.append(f"summary.losses={s.get('losses')} != events_losses={loss_ev}") + # run.toml totals + try: + import tomllib + with open(run_dir / "run.toml", "rb") as f: + t = tomllib.load(f) + if t.get("run", {}).get("total_episodes") != len(valid): + errs.append(f"run.toml.total_episodes != {len(valid)}") + if t.get("run", {}).get("total_steps") != steps_ev: + errs.append(f"run.toml.total_steps != {steps_ev}") + except Exception as e: + errs.append(f"run.toml parse fail: {e!r}") + return errs + + @staticmethod + def _atomic_promote(candidate_dir: Path, live_dir: Path) -> None: + """Directory-level atomic promotion: candidate → live. + + Two-rename transaction on the SAME filesystem (guaranteed by staging + under ``/.staging/``): + + 1. (recovery) If backup exists but live doesn't → crash interrupted + between step 3 and 4 → restore backup → live. + 2. (recovery) If both live and backup exist → crash interrupted + after step 4 but before cleanup → promotion succeeded, delete backup. + 3. If live exists: ``os.rename(live, backup)`` → live now absent. + 4. ``os.rename(candidate, live)`` → candidate now at live path. + 5. If step 4 raised: ``os.rename(backup, live)`` → restore old. + 6. If step 4 succeeded: ``shutil.rmtree(backup)``. + + Uses ``os.rename`` (not ``os.replace``) because on Windows renaming + to a NON-existent destination works for directories, whereas + ``os.replace`` on a non-empty dir raises WinError 5. We guarantee + the destination doesn't exist by renaming live→backup first. + """ + # The implementation below is deliberately directory-granular: the + # three publishable files move together, never one at a time. Keep + # the legacy implementation below unreachable for compatibility with + # older patches; all callers return through this transaction path. + import shutil as _sh + from agentbench_frame.games.miracle.atomicio import atomic_write_json + if not candidate_dir.exists(): + raise FileNotFoundError(f"candidate missing: {candidate_dir}") + backup_dir = live_dir.with_name(live_dir.name + ".backup_promote") + marker_path = live_dir.with_name(live_dir.name + ".promote_marker.json") + MatrixRunner._recover_promotion_transaction(live_dir, backup_dir, marker_path) + live_dir.parent.mkdir(parents=True, exist_ok=True) + atomic_write_json(marker_path, { + "schema_version": 1, + "state": "prepared", + "live_name": live_dir.name, + "backup_name": backup_dir.name, + "candidate_name": candidate_dir.name, + }) + prior_live = live_dir.exists() + if prior_live: + if backup_dir.exists(): + raise RuntimeError(f"refusing to overwrite promotion backup: {backup_dir}") + os.rename(str(live_dir), str(backup_dir)) + # Test-only seam: a real child process exits after the dangerous + # first rename. It is never enabled by normal callers. + if os.environ.get("MIRACLE_TEST_KILL_AFTER_LIVE_RENAME") == "1": + os._exit(86) + try: + os.rename(str(candidate_dir), str(live_dir)) + except Exception: + if prior_live and backup_dir.exists(): + if live_dir.exists(): + preserved = live_dir.with_name(live_dir.name + ".failed_candidate") + if preserved.exists(): + raise RuntimeError(f"refusing to overwrite {preserved}") + os.rename(str(live_dir), str(preserved)) + os.rename(str(backup_dir), str(live_dir)) + raise + if prior_live and backup_dir.exists(): + _sh.rmtree(backup_dir) + try: + marker_path.unlink() + except FileNotFoundError: + pass + return + + import shutil as _sh + if not candidate_dir.exists(): + raise FileNotFoundError(f"candidate missing: {candidate_dir}") + backup_dir = live_dir.with_name(live_dir.name + ".backup_promote") + + # --- crash recovery --- + MatrixRunner._recover_interrupted_promote(live_dir, backup_dir) + + live_parent = live_dir.parent + live_parent.mkdir(parents=True, exist_ok=True) + + had_existing = live_dir.exists() and any(live_dir.iterdir()) + + if had_existing: + # Step 3: rename live → backup (live now absent) + if backup_dir.exists(): + _sh.rmtree(backup_dir) + os.rename(str(live_dir), str(backup_dir)) + else: + # live doesn't exist or is empty — remove if empty + if live_dir.exists(): + _sh.rmtree(live_dir) + backup_dir.mkdir(exist_ok=True) # empty placeholder for rollback safety + + try: + # Step 4: rename candidate → live (candidate now at live path) + os.rename(str(candidate_dir), str(live_dir)) + except Exception: + # Step 5: restore backup → live + if backup_dir.exists() and backup_dir != live_dir: + try: + if live_dir.exists(): + _sh.rmtree(live_dir) + os.rename(str(backup_dir), str(live_dir)) + except Exception: + pass # best-effort; caller sees the original exception + raise + + # Step 6: success — clean backup + if backup_dir.exists(): + try: + _sh.rmtree(backup_dir) + except Exception: + pass # non-critical; leftover is detected on next run + + @staticmethod + def _recover_promotion_transaction(live_dir: Path, backup_dir: Path, + marker_path: Path) -> None: + """Conservatively recover a directory promotion interrupted by a kill. + + A marker denotes an incomplete transaction. When both the old backup + and a candidate at the live path exist, the old live bytes win: the + candidate is moved aside for inspection and is never published by + inference. This is intentionally stricter than treating both paths + as a successful promotion. + """ + backup_exists = backup_dir.exists() + live_exists = live_dir.exists() + if not marker_path.exists(): + if backup_exists: + raise RuntimeError( + f"refusing ambiguous promotion state without marker: {backup_dir}" + ) + return + if backup_exists and not live_exists: + os.rename(str(backup_dir), str(live_dir)) + elif backup_exists and live_exists: + preserved = live_dir.with_name(live_dir.name + ".interrupted_candidate") + if preserved.exists(): + raise RuntimeError(f"refusing to overwrite {preserved}") + os.rename(str(live_dir), str(preserved)) + os.rename(str(backup_dir), str(live_dir)) + # A marker without a backup is a first-write transaction. The live + # directory is its only complete copy; leave it untouched. + if live_dir.exists(): + try: + marker_path.unlink() + except FileNotFoundError: + pass + + @staticmethod + def _recover_interrupted_promote(live_dir: Path, backup_dir: Path) -> None: + """Detect and recover from a crash during a previous _atomic_promote. + + Invariants after a clean run: live exists, backup does NOT exist. + Crash states: + - backup exists, live absent: interrupted between step 3 and 4 + → restore backup → live. + - both exist: interrupted after step 4 but before cleanup + → promotion succeeded; delete backup. + """ + import shutil as _sh + b_exists = backup_dir.exists() + l_exists = live_dir.exists() + if b_exists and not l_exists: + # crash between rename(live→backup) and rename(candidate→live) + try: + os.rename(str(backup_dir), str(live_dir)) + except OSError: + pass # best-effort + elif b_exists and l_exists: + # crash after successful promotion, before cleanup + try: + _sh.rmtree(backup_dir) + except Exception: + pass + + @staticmethod + def _cleanup_staging(staging_root: Path, run_id: str) -> None: + """Clean any leftover staging entries for this run_id.""" + import shutil as _sh + candidate = staging_root / run_id + if candidate.exists(): + try: + _sh.rmtree(candidate) + except Exception: + pass + backup = staging_root / (run_id + ".backup_promote") + if backup.exists(): + try: + _sh.rmtree(backup) + except Exception: + pass diff --git a/tests/miracle/test_atomic_windows_retry.py b/tests/miracle/test_atomic_windows_retry.py new file mode 100644 index 0000000..2fd4f27 --- /dev/null +++ b/tests/miracle/test_atomic_windows_retry.py @@ -0,0 +1,39 @@ +import hashlib, json +from pathlib import Path +import pytest + +def _winerror(): + e = PermissionError(5, "sharing violation") + e.winerror = 5 + return e + +def test_unique_temp_and_transient_windows_retry(tmp_path, monkeypatch): + from agentbench_frame.games.miracle import atomicio as aio + target = tmp_path / "progress.json"; calls=[]; real=aio.os.replace + monkeypatch.setattr(aio, "_is_windows", lambda: True) + def flaky(src, dst): + calls.append(Path(src).name) + if len(calls) < 3: raise _winerror() + return real(src, dst) + monkeypatch.setattr(aio.os, "replace", flaky) + aio.atomic_write_json(target, {"ok": True}) + assert len(calls) == 3 and calls[0] == calls[1] == calls[2] + assert calls[0] != "progress.json.tmp" + assert json.loads(target.read_text(encoding="utf-8")) == {"ok": True} + +def test_persistent_windows_permission_preserves_old_target(tmp_path, monkeypatch): + from agentbench_frame.games.miracle import atomicio as aio + target = tmp_path / "progress.json"; target.write_text('{"old":true}', encoding="utf-8") + before=hashlib.sha256(target.read_bytes()).hexdigest(); monkeypatch.setattr(aio, "_is_windows", lambda: True) + monkeypatch.setattr(aio.os, "replace", lambda *a: (_ for _ in ()).throw(_winerror())) + with pytest.raises(PermissionError): aio.atomic_write_json(target, {"new": True}) + assert hashlib.sha256(target.read_bytes()).hexdigest() == before + assert json.loads(target.read_text(encoding="utf-8")) == {"old": True} + assert list(tmp_path.glob(".progress.json.*.tmp")) + +def test_progress_uses_shared_unique_writer(tmp_path): + from agentbench_frame.games.miracle.matrix import write_progress_atomic + a=tmp_path/"a.json"; b=tmp_path/"b.json" + write_progress_atomic(a,{"attempts":{"a":1}}); write_progress_atomic(b,{"attempts":{"b":2}}) + assert json.loads(a.read_text()) != json.loads(b.read_text()) + assert not (tmp_path/"a.json.tmp").exists() and not (tmp_path/"b.json.tmp").exists() diff --git a/tests/miracle/test_matrix.py b/tests/miracle/test_matrix.py new file mode 100644 index 0000000..e2296b2 --- /dev/null +++ b/tests/miracle/test_matrix.py @@ -0,0 +1,223 @@ +"""Red-light tests for the 32-game matrix orchestrator (阶段正式矩阵 §3 TDD gate). + +Tests MUST pass before the first real game starts. They cover the spec's safety +requirements using synthesized MatchAttempt-like objects + tmp progress files — +no Judge, no real match. +""" +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from agentbench_frame.games.miracle import matrix # red-light import +from agentbench_frame.games.miracle.matrix import ( + aggregate, + classify_game, + is_done, + load_progress, + make_attempt_plan, + mark_done, + next_incomplete, + rank_audit, + should_stop, + write_progress_atomic, +) + + +# ---- fake MatchAttempt (same shape as match_runner.MatchAttempt for classify_game) ---- # +class FakeAtt: + def __init__(self, *, game_id="g", rank=None, camp=None, valid=True, normalized_result="win", + raw_winner=None, error_type=None, wrapper_timeout=False, + result_json_status="ok", discrepancies=None, realized_randomization=None, + scores=None, steps=10, ai_crash_player=None, judge_crash=False, + cleanup_procs=None, evidence_paths=None): + self.game_id = game_id; self.rank = rank; self.camp = camp + self.valid = valid; self.normalized_result = normalized_result + self.raw_winner = raw_winner if raw_winner is not None else ( + None if camp is None else (camp if normalized_result == "win" else 1 - camp)) + self.error_type = error_type; self.wrapper_timeout = wrapper_timeout + self.result_json_status = result_json_status + self.discrepancies = discrepancies or [] + self.realized_randomization = realized_randomization or {"map_type": 0, "day_time": 1} + self.scores = scores or {"0": 5, "1": 2}; self.steps = steps + self.ai_crash_player = ai_crash_player; self.judge_crash = judge_crash + self.winner_agent = "miracle_ifelse" if normalized_result == "win" else "rank" + self.normal_cleanup_nonzero = False; self.reason = error_type or "" + self.evidence_paths = evidence_paths or {"result_json": ""} + + +# ---- 1. exactly 32 unique attempts ---- # +def test_plan_has_32_unique_attempts(): + plan = make_attempt_plan() + ids = [a["game_id"] for a in plan] + assert len(plan) == 32 + assert len(set(ids)) == 32 + + +# ---- 2. each rank camp0/camp1 once ---- # +def test_plan_each_rank_both_camps_once(): + plan = make_attempt_plan() + for rank in range(1, 17): + camps = sorted(a["camp"] for a in plan if a["rank"] == rank) + assert camps == [0, 1], f"rank{rank} camps={camps}" + + +def test_plan_order_is_rank_then_camp(): + plan = make_attempt_plan() + seq = [(a["rank"], a["camp"]) for a in plan] + assert seq[0] == (1, 0) and seq[1] == (1, 1) and seq[-1] == (16, 1) + + +# ---- 3. completed never rerun ---- # +def test_next_incomplete_skips_done(): + plan = make_attempt_plan() + prog = {"attempts": {}} + first = next_incomplete(prog, plan) + assert first["game_id"] == plan[0]["game_id"] + mark_done(prog, plan[0]["game_id"], {"valid": True, "normalized_result": "win"}) + assert is_done(prog, plan[0]["game_id"]) is True + nxt = next_incomplete(prog, plan) + assert nxt["game_id"] == plan[1]["game_id"] + + +def test_next_incomplete_none_when_all_done(): + plan = make_attempt_plan() + prog = {"attempts": {}} + for a in plan: + mark_done(prog, a["game_id"], {"valid": True}) + assert next_incomplete(prog, plan) is None + + +# ---- 4. session exists not overwrite/delete (reuses smoke_audit) ---- # +def test_fresh_session_refused_if_exists(tmp_path): + from agentbench_frame.games.miracle.smoke_audit import ensure_fresh_session + sid = matrix.make_session_id() + sd = ensure_fresh_session(tmp_path, sid) + (sd / "x").write_text("keep") + with pytest.raises(FileExistsError): + ensure_fresh_session(tmp_path, sid) + assert (sd / "x").read_text() == "keep" + + +# ---- 5. progress atomic write ---- # +def test_progress_atomic_write(tmp_path): + prog = {"session_id": "s1", "attempts": {"g1": {"state": "done"}}} + p = tmp_path / "progress.json" + write_progress_atomic(p, prog) + loaded = json.loads(p.read_text(encoding="utf-8")) + assert loaded == prog + assert not (tmp_path / "progress.json.tmp").exists() + + +def test_progress_load_handles_missing(tmp_path): + assert load_progress(tmp_path / "nope.json") == {"attempts": {}} + + +# ---- 6. result append no dup (events append) ---- # +def test_append_event_no_duplicate(tmp_path): + evp = tmp_path / "events.jsonl" + matrix.append_event_atomic(evp, {"event": "game", "game_id": "g1"}) + matrix.append_event_atomic(evp, {"event": "game", "game_id": "g1"}) + matrix.append_event_atomic(evp, {"event": "game", "game_id": "g2"}) + lines = [l for l in evp.read_text(encoding="utf-8").splitlines() if l.strip()] + ids = [json.loads(l)["game_id"] for l in lines] + # no duplicate game_id entries + assert len(ids) == len(set(ids)) == 2 + + +# ---- 7 + 8. classify: valid/invalid; AI crash is NOT a valid win ---- # +def test_classify_valid_win_camp0(): + rec = classify_game(FakeAtt(game_id="m_rank01_camp0", rank=1, camp=0, normalized_result="win", raw_winner=0)) + assert rec["valid"] is True and rec["normalized_result"] == "win" + assert rec["rank"] == 1 and rec["camp"] == 0 + + +def test_classify_ai_crash_not_valid_win(): + # Judge may award ifelse the win (raw_winner=camp), but AI crash => invalid, not a capability win + att = FakeAtt(game_id="m_rank03_camp0", rank=3, camp=0, normalized_result="error", + error_type="ai_crash", ai_crash_player=1, raw_winner=0, valid=False) + rec = classify_game(att) + assert rec["valid"] is False + assert rec["normalized_result"] != "win" + + +# ---- 10. camp-swap winner normalization ---- # +def test_classify_camp_swap_normalization(): + # camp0 raw_winner=0 -> ifelse win; camp1 raw_winner=0 -> ifelse loss + r0 = classify_game(FakeAtt(game_id="a", rank=1, camp=0, normalized_result="win", raw_winner=0)) + r1 = classify_game(FakeAtt(game_id="b", rank=1, camp=1, normalized_result="loss", raw_winner=0)) + assert r0["normalized_result"] == "win" and r1["normalized_result"] == "loss" + assert r0["ifelse_camp"] == 0 and r1["ifelse_camp"] == 1 + + +# ---- 11. recovery audit distinguishes not_started/running/done ---- # +def test_progress_states_distinguished(): + plan = make_attempt_plan() + prog = {"attempts": {}} + assert matrix.state_of(prog, plan[0]["game_id"]) == "not_started" + matrix.mark_running(prog, plan[0]["game_id"]) + assert matrix.state_of(prog, plan[0]["game_id"]) == "running" + matrix.mark_done(prog, plan[0]["game_id"], {"valid": True}) + assert matrix.state_of(prog, plan[0]["game_id"]) == "done" + + +# ---- 9 + 13. infra error blocks next batch; success/invalid/infra distinct ---- # +def test_should_stop_on_infra_anomalies(): + assert should_stop(FakeAtt(wrapper_timeout=True)) == ("stop", "wrapper_timeout") + assert should_stop(FakeAtt(error_type="evidence_mismatch"))[0] == "stop" + assert should_stop(FakeAtt(result_json_status="missing"))[0] == "stop" + # a clean valid game does NOT stop + assert should_stop(FakeAtt(normalized_result="win"))[0] != "stop" + # an AI-crash invalid does NOT stop the matrix (recorded, continue) + assert should_stop(FakeAtt(normalized_result="error", error_type="ai_crash"))[0] != "stop" + + +# ---- rank audit ---- # +def test_rank_audit_pass_for_two_clean_games(): + games = [FakeAtt(game_id="m_rank01_camp0", rank=1, camp=0, normalized_result="win"), + FakeAtt(game_id="m_rank01_camp1", rank=1, camp=1, normalized_result="loss")] + ok, reasons = rank_audit(1, games) + assert ok is True, reasons + + +def test_rank_audit_fail_on_missing_camp(): + games = [FakeAtt(game_id="m_rank01_camp0", rank=1, camp=0, normalized_result="win"), + FakeAtt(game_id="m_rank01_camp0b", rank=1, camp=0, normalized_result="win")] # both camp0 + ok, reasons = rank_audit(1, games) + assert ok is False + + +# ---- 12. PID residual check wired into stop ---- # +def test_residual_proc_triggers_stop(tmp_path): + # write a result-json with a fake live-managed-proc identity pointing at this process + import os, psutil + me = psutil.Process(os.getpid()) + rj = tmp_path / "r.json" + rj.write_text(json.dumps({"judge": {"pid": os.getpid(), "started_at": me.create_time(), "role": "judge"}})) + att = FakeAtt(evidence_paths={"result_json": str(rj)}) + stop, reason = matrix.should_stop_with_residual(att) + assert stop == "stop" and "residual" in reason + + +# ---- aggregate ---- # +def test_aggregate_counts_and_winrate(): + recs = [] + for rank in range(1, 4): + recs.append({"rank": rank, "camp": 0, "valid": True, "normalized_result": "win", "raw_winner": 0, "steps": 40, "realized_randomization": {"map_type": 0, "day_time": 1}, "scores": {"0": 5, "1": 2}, "error_type": None}) + recs.append({"rank": rank, "camp": 1, "valid": True, "normalized_result": "loss", "raw_winner": 0, "steps": 38, "realized_randomization": {"map_type": 1, "day_time": 0}, "scores": {"0": 5, "1": 2}, "error_type": None}) + recs.append({"rank": 3, "camp": 0, "valid": False, "normalized_result": "error", "raw_winner": None, "steps": 5, "realized_randomization": None, "scores": None, "error_type": "ai_crash"}) + agg = aggregate(recs) + assert agg["total_attempts"] == 7 + assert agg["valid_games"] == 6 and agg["invalid_games"] == 1 + assert agg["wins"] == 3 and agg["losses"] == 3 + assert agg["win_rate"] == pytest.approx(0.5) + assert agg["per_rank"][1]["win_rate"] == pytest.approx(0.5) # rank1: camp0 win + camp1 loss + assert agg["per_rank"][3]["invalid"] == 1 + + +def test_aggregate_no_valid_winrate_null(): + recs = [{"rank": 1, "camp": 0, "valid": False, "normalized_result": "error", "raw_winner": None, "steps": 1, "realized_randomization": None, "scores": None, "error_type": "ai_crash"}] + agg = aggregate(recs) + assert agg["win_rate"] is None diff --git a/tests/miracle/test_matrix_cli_flow.py b/tests/miracle/test_matrix_cli_flow.py new file mode 100644 index 0000000..1a242c5 --- /dev/null +++ b/tests/miracle/test_matrix_cli_flow.py @@ -0,0 +1,247 @@ +from __future__ import annotations + +import importlib +import json +import sys +from pathlib import Path + +import pytest + + +def _import_cli_module(monkeypatch): + repo = Path(__file__).resolve().parents[2] + tools_dir = str(repo / "tools") + if tools_dir not in sys.path: + sys.path.insert(0, tools_dir) + monkeypatch.setenv("AGENTBENCH_ROOT", str(repo)) + monkeypatch.setenv("MIRACLE_IFELSE_DIR", str(repo)) + sys.modules.pop("miracle_matrix", None) + return importlib.import_module("miracle_matrix") + + +def test_main_missing_protocol_returns_preflight_error_without_session( + tmp_path, monkeypatch, capsys +): + mm = _import_cli_module(monkeypatch) + rc = mm.main([ + "--dry-run", + "--protocol", str(tmp_path / "missing.json"), + "--roster", str(tmp_path / "roster.json"), + "--session-root", str(tmp_path / "sessions"), + ]) + + assert rc == 2 + assert "protocol file missing" in capsys.readouterr().err + assert not (tmp_path / "sessions").exists() + + +def test_main_missing_explicit_asset_stops_before_runner_or_session(tmp_path, monkeypatch, capsys): + mm = _import_cli_module(monkeypatch) + invoked = [] + monkeypatch.setattr(mm, "MatrixRunner", lambda **_kwargs: invoked.append(True)) + rc = mm.main(["--dry-run", "--judge-dir", str(tmp_path / "missing-judge"), + "--session-root", str(tmp_path / "sessions")]) + assert rc == 2 + assert "judge directory missing" in capsys.readouterr().err + assert invoked == [] + assert not (tmp_path / "sessions").exists() + + +def test_cli_resume_verifier_receives_control_hashes(tmp_path, monkeypatch): + mm = _import_cli_module(monkeypatch) + sid = "sid" + session_dir = tmp_path / sid + session_dir.mkdir() + monkeypatch.setattr(mm, "SESSION_ROOT", tmp_path) + observed = {} + + def fake_verify(_session_dir, **kwargs): + observed.update(kwargs) + return False, ["stop"] + + monkeypatch.setattr(mm, "verify_session_for_resume", fake_verify) + + class FakeRunner: + def __init__(self): + self.session_dir = session_dir + + control_inputs = mm.ControlInputs( + protocol=json.loads(mm.PROTOCOL.read_text(encoding="utf-8")), + roster=json.loads(mm.ROSTER.read_text(encoding="utf-8")), + hashes={"protocol": "p", "roster": "r"}, + ) + roots = {name: tmp_path / name for name in ( + "judge_dir", "ifelse_dir", "extracted_root", "archives_root", + "precheck_root", "rank16_build_root", + )} + rc = mm._run_resume(FakeRunner(), sid, control_inputs=control_inputs, **roots) + + assert rc == 2 + assert observed["expected_control_inputs"] == {"protocol": "p", "roster": "r"} + + +def test_main_manifest_hashes_explicit_judge_and_ifelse_dirs(tmp_path, monkeypatch): + mm = _import_cli_module(monkeypatch) + explicit_ifelse = tmp_path / "ifelse"; explicit_ifelse.mkdir() + explicit_judge = tmp_path / "judge"; explicit_judge.mkdir() + (explicit_ifelse / "main.py").write_text("ifelse", encoding="utf-8") + (explicit_judge / "main.py").write_text("judge", encoding="utf-8") + missing = tmp_path / "missing-global" + monkeypatch.setattr(mm, "IFELSE", missing) + monkeypatch.setattr(mm, "JUDGE", missing) + monkeypatch.setattr(mm, "validate_runtime_paths", lambda **_: None) + def verified_hashes(*_args, asset_digests, **_kwargs): + asset_digests.update({ + "ifelse": mm.sha(explicit_ifelse / "main.py"), + "judge": mm.sha(explicit_judge / "main.py"), + }) + return [] + monkeypatch.setattr(mm, "verify_hashes", verified_hashes) + seen = {} + + class FakeRunner: + def __init__(self, **_kwargs): + self.session_id = "fake"; self.run_id = "run"; self.session_dir = tmp_path / "session" + self.timeout = 8.0; self.wrapper_timeout_s = 180.0 + def prepare_session(self): self.session_dir.mkdir() + def record_manifest(self, **kwargs): seen.update(kwargs) + def dry_run(self): return {"plan_count": 1, "plan": [{"game_id": "first"}]} + + monkeypatch.setattr(mm, "MatrixRunner", FakeRunner) + assert mm.main(["--dry-run", "--ifelse-dir", str(explicit_ifelse), + "--judge-dir", str(explicit_judge)]) == 0 + assert seen["ifelse_sha"] == mm.sha(explicit_ifelse / "main.py") + assert seen["judge_sha"] == mm.sha(explicit_judge / "main.py") + + +def _make_main_assets(tmp_path, mm): + roots = {name: tmp_path / name for name in ( + "judge", "ifelse", "extracted", "archives", "precheck", "rank16", + )} + for root in roots.values(): + root.mkdir() + (roots["judge"] / "main.py").write_text("judge\n", encoding="utf-8") + (roots["ifelse"] / "main.py").write_text("ifelse\n", encoding="utf-8") + strategies = [] + builds = {} + archives = {} + runnables = {} + for rank in range(1, 17): + archive = roots["archives"] / f"rank{rank:02d}__fixture.zip" + archive.write_bytes(f"archive-{rank}".encode("ascii")) + archives[rank] = archive + strategy = {"rank": rank, "archive_sha256": mm.sha(archive)} + if rank in mm.PYTHON_RANKS: + directory = roots["extracted"] / f"rank{rank:02d}__fixture" + directory.mkdir() + runnable = directory / "main.py" + runnable.write_text(f"python-{rank}\n", encoding="utf-8") + runnables[rank] = runnable + strategy.update({"entry": "main.py", "runnable_sha256": mm.sha(runnable)}) + else: + directory = (roots["rank16"] / "rank16_copy" if rank == 16 + else roots["precheck"] / "strategies" / f"rank{rank:02d}") + directory.mkdir(parents=True) + executable = directory / "main.exe" + executable.write_bytes(f"exe-{rank}".encode("ascii")) + runnables[rank] = executable + builds[f"rank{rank:02d}"] = mm.sha(executable) + strategies.append(strategy) + protocol = { + "frozen_identities": { + "evaluated_agent": {"sha256": mm.sha(roots["ifelse"] / "main.py")}, + "judge": {"main_py_sha256": mm.sha(roots["judge"] / "main.py")}, + "build_artifacts_win64_mingw": builds, + } + } + protocol_path = tmp_path / "protocol.json" + roster_path = tmp_path / "roster.json" + protocol_path.write_text(json.dumps(protocol), encoding="utf-8") + roster_path.write_text(json.dumps({"strategies": strategies}), encoding="utf-8") + return roots, archives, runnables, protocol_path, roster_path + + +@pytest.mark.parametrize("asset", ["judge", "ifelse", "archive", "python", "cpp"]) +def test_main_unreadable_asset_stops_before_manifest_or_execution(tmp_path, monkeypatch, capsys, asset): + mm = _import_cli_module(monkeypatch) + roots, archives, runnables, protocol_path, roster_path = _make_main_assets(tmp_path, mm) + unreadable = { + "judge": roots["judge"] / "main.py", + "ifelse": roots["ifelse"] / "main.py", + "archive": archives[1], + "python": runnables[4], + "cpp": runnables[1], + }[asset] + original_sha = mm.sha + + def injected_sha(path): + if Path(path) == unreadable: + raise OSError("injected unreadable asset") + return original_sha(path) + + monkeypatch.setattr(mm, "sha", injected_sha) + recorded = [] + dry_runs = [] + executes = [] + session_root = tmp_path / "sessions" + + class FakeRunner: + def __init__(self, **_kwargs): + self.session_dir = session_root / "fake" + self.session_id = "fake" + self.run_id = "run" + + def prepare_session(self): + self.session_dir.mkdir(parents=True) + + def record_manifest(self, **_kwargs): + recorded.append(True) + + def dry_run(self): + dry_runs.append(True) + + def execute(self): + executes.append(True) + + monkeypatch.setattr(mm, "MatrixRunner", FakeRunner) + rc = mm.main([ + "--dry-run", "--protocol", str(protocol_path), "--roster", str(roster_path), + "--session-root", str(session_root), "--judge-dir", str(roots["judge"]), + "--ifelse-dir", str(roots["ifelse"]), "--extracted-root", str(roots["extracted"]), + "--archives-root", str(roots["archives"]), "--precheck-root", str(roots["precheck"]), + "--rank16-build-root", str(roots["rank16"]), + ]) + + assert rc == 2 + assert "FATAL: hash mismatches" in capsys.readouterr().out + assert recorded == [] + assert dry_runs == [] + assert executes == [] + + +def test_resume_current_asset_mismatch_stops_before_log_or_resume(tmp_path, monkeypatch): + mm = _import_cli_module(monkeypatch) + sid = "sid"; session = tmp_path / sid; session.mkdir() + monkeypatch.setattr(mm, "verify_session_for_resume", lambda *_a, **_k: (True, [])) + roots = {name: tmp_path / name for name in ("judge", "ifelse", "extracted", "archives", "precheck", "rank16")} + seen = {} + def mismatch(*_args, **kwargs): + seen.update(kwargs); return ["judge sha mismatch"] + monkeypatch.setattr(mm, "verify_hashes", mismatch) + class FakeRunner: + session_dir = session + def resume(self, _sid): raise AssertionError("resume must not run") + control = mm.ControlInputs( + json.loads(mm.PROTOCOL.read_text(encoding="utf-8")), + json.loads(mm.ROSTER.read_text(encoding="utf-8")), + {"protocol": "p", "roster": "r"}, + ) + before = {p.name: p.read_bytes() for p in session.iterdir()} + rc = mm._run_resume(FakeRunner(), sid, control_inputs=control, session_root=tmp_path, + judge_dir=roots["judge"], ifelse_dir=roots["ifelse"], + extracted_root=roots["extracted"], archives_root=roots["archives"], + precheck_root=roots["precheck"], rank16_build_root=roots["rank16"]) + assert rc == 2 + assert seen["judge_root"] == roots["judge"] + assert not (session / "matrix.full.log").exists() + assert before == {p.name: p.read_bytes() for p in session.iterdir()} diff --git a/tests/miracle/test_matrix_cli_inputs.py b/tests/miracle/test_matrix_cli_inputs.py new file mode 100644 index 0000000..1df8e01 --- /dev/null +++ b/tests/miracle/test_matrix_cli_inputs.py @@ -0,0 +1,278 @@ +from __future__ import annotations + +import importlib +import json +import sys +from pathlib import Path + +import pytest + + +def _valid_protocol(mm): + return { + "frozen_identities": { + "evaluated_agent": {"sha256": "a" * 64}, + "judge": {"main_py_sha256": "b" * 64}, + "build_artifacts_win64_mingw": { + f"rank{rank:02d}": "c" * 64 for rank in mm.CPP_RANKS + }, + } + } + + +def _valid_roster(mm): + strategies = [] + for rank in range(1, 17): + strategy = {"rank": rank, "archive_sha256": "d" * 64} + if rank in mm.PYTHON_RANKS: + strategy.update({"entry": "main.py", "runnable_sha256": "e" * 64}) + strategies.append(strategy) + return {"strategies": strategies} + + +def _write_controls(protocol_path, roster_path, protocol, roster): + protocol_path.write_text(json.dumps(protocol), encoding="utf-8") + roster_path.write_text(json.dumps(roster), encoding="utf-8") + + +def _import_cli_module(monkeypatch): + repo = Path(__file__).resolve().parents[2] + tools_dir = str(repo / "tools") + if tools_dir not in sys.path: + sys.path.insert(0, tools_dir) + monkeypatch.setenv("AGENTBENCH_ROOT", str(repo)) + monkeypatch.setenv("MIRACLE_IFELSE_DIR", str(repo)) + sys.modules.pop("miracle_matrix", None) + return importlib.import_module("miracle_matrix") + + +def test_load_control_inputs_reports_missing_file(tmp_path, monkeypatch): + mm = _import_cli_module(monkeypatch) + + with pytest.raises(mm.PreflightError, match="protocol file missing"): + mm.load_control_inputs(tmp_path / "missing.json", tmp_path / "roster.json") + + +def test_parse_args_accepts_explicit_control_paths(tmp_path, monkeypatch): + mm = _import_cli_module(monkeypatch) + args = mm.parse_args([ + "--dry-run", + "--protocol", str(tmp_path / "p.json"), + "--roster", str(tmp_path / "r.json"), + ]) + + assert args.dry_run is True + assert args.protocol == tmp_path / "p.json" + assert args.roster == tmp_path / "r.json" + + +def test_load_control_inputs_validates_protocol_and_roster(tmp_path, monkeypatch): + mm = _import_cli_module(monkeypatch) + protocol = _valid_protocol(mm) + roster = _valid_roster(mm) + protocol_path = tmp_path / "protocol.json" + roster_path = tmp_path / "roster.json" + protocol_path.write_text(json.dumps(protocol), encoding="utf-8") + roster_path.write_text(json.dumps(roster), encoding="utf-8") + + inputs = mm.load_control_inputs(protocol_path, roster_path) + + assert inputs.protocol == protocol + assert inputs.roster == roster + assert set(inputs.hashes) == {"protocol", "roster"} + assert len(inputs.hashes["protocol"]) == 64 + assert len(inputs.hashes["roster"]) == 64 + + +@pytest.mark.parametrize( + "missing_field", + [ + "frozen_identities", + "evaluated_agent_sha", + "judge_sha", + *[f"cpp_sha:{rank}" for rank in (1, 2, 3, 6, 8, 9, 10, 11, 12, 13, 14, 15, 16)], + *[f"archive_sha:{rank}" for rank in range(1, 17)], + *[f"python_entry:{rank}" for rank in (4, 5, 7)], + *[f"python_runnable_sha:{rank}" for rank in (4, 5, 7)], + ], +) +def test_main_rejects_incomplete_control_schema_before_session_write( + tmp_path, monkeypatch, capsys, missing_field +): + mm = _import_cli_module(monkeypatch) + protocol = _valid_protocol(mm) + roster = _valid_roster(mm) + identities = protocol["frozen_identities"] + if missing_field == "frozen_identities": + protocol.pop("frozen_identities") + elif missing_field == "evaluated_agent_sha": + identities["evaluated_agent"].pop("sha256") + elif missing_field == "judge_sha": + identities["judge"].pop("main_py_sha256") + elif missing_field.startswith("cpp_sha:"): + rank = int(missing_field.split(":", 1)[1]) + identities["build_artifacts_win64_mingw"].pop(f"rank{rank:02d}") + elif missing_field.startswith("archive_sha:"): + rank = int(missing_field.split(":", 1)[1]) + roster["strategies"][rank - 1].pop("archive_sha256") + elif missing_field.startswith("python_entry:"): + rank = int(missing_field.split(":", 1)[1]) + roster["strategies"][rank - 1].pop("entry") + else: + rank = int(missing_field.split(":", 1)[1]) + roster["strategies"][rank - 1].pop("runnable_sha256") + + protocol_path = tmp_path / "protocol.json" + roster_path = tmp_path / "roster.json" + session_root = tmp_path / "sessions" + _write_controls(protocol_path, roster_path, protocol, roster) + constructed = [] + prepared = [] + + class FakeRunner: + def __init__(self, **_kwargs): + constructed.append(True) + self.session_dir = session_root / "unexpected" + self.session_id = "unexpected" + self.run_id = "unexpected" + + def prepare_session(self): + prepared.append(True) + self.session_dir.mkdir(parents=True) + + monkeypatch.setattr(mm, "validate_runtime_paths", lambda **_kwargs: None) + monkeypatch.setattr(mm, "MatrixRunner", FakeRunner) + + rc = mm.main([ + "--dry-run", "--protocol", str(protocol_path), "--roster", str(roster_path), + "--session-root", str(session_root), + ]) + + assert rc == 2 + assert "FATAL:" in capsys.readouterr().err + assert constructed == [] + assert prepared == [] + assert not session_root.exists() + assert not (session_root / "unexpected" / "matrix.full.log").exists() + + +@pytest.mark.parametrize("field", [ + "evaluated_agent", "judge", "cpp_build", "archive", "python_runnable", +]) +@pytest.mark.parametrize("malformed_sha", [ + "a" * 63, "a" * 65, "g" * 64, "A" * 64, " " + "a" * 64 + " ", None, +]) +def test_main_rejects_malformed_control_sha_before_session_write( + tmp_path, monkeypatch, capsys, field, malformed_sha +): + mm = _import_cli_module(monkeypatch) + protocol = _valid_protocol(mm) + roster = _valid_roster(mm) + if field == "evaluated_agent": + protocol["frozen_identities"]["evaluated_agent"]["sha256"] = malformed_sha + elif field == "judge": + protocol["frozen_identities"]["judge"]["main_py_sha256"] = malformed_sha + elif field == "cpp_build": + protocol["frozen_identities"]["build_artifacts_win64_mingw"]["rank01"] = malformed_sha + elif field == "archive": + roster["strategies"][0]["archive_sha256"] = malformed_sha + else: + roster["strategies"][3]["runnable_sha256"] = malformed_sha + + protocol_path = tmp_path / "protocol.json" + roster_path = tmp_path / "roster.json" + session_root = tmp_path / "sessions" + _write_controls(protocol_path, roster_path, protocol, roster) + constructed = [] + prepared = [] + + class FakeRunner: + def __init__(self, **_kwargs): + constructed.append(True) + + def prepare_session(self): + prepared.append(True) + + monkeypatch.setattr(mm, "validate_runtime_paths", lambda **_kwargs: None) + monkeypatch.setattr(mm, "MatrixRunner", FakeRunner) + + rc = mm.main([ + "--dry-run", "--protocol", str(protocol_path), "--roster", str(roster_path), + "--session-root", str(session_root), + ]) + + assert rc == 2 + assert "FATAL: control schema" in capsys.readouterr().err + assert constructed == [] + assert prepared == [] + assert not session_root.exists() + + +def test_control_text_sha_is_identical_for_lf_and_crlf(tmp_path, monkeypatch): + mm = _import_cli_module(monkeypatch) + lf = tmp_path / "lf.json" + crlf = tmp_path / "crlf.json" + lf.write_bytes(b'{\n "protocol_version": "test"\n}\n') + crlf.write_bytes(b'{\r\n "protocol_version": "test"\r\n}\r\n') + assert mm.control_text_sha(lf) == mm.control_text_sha(crlf) + + +def test_windows_crlf_protocol_checkout_passes_expected_hash(tmp_path, monkeypatch): + mm = _import_cli_module(monkeypatch) + protocol = tmp_path / "protocol.json" + roster = tmp_path / "roster.json" + protocol.write_bytes(json.dumps(_valid_protocol(mm), indent=2).replace("\n", "\r\n").encode("utf-8")) + roster.write_text(json.dumps(_valid_roster(mm)), encoding="utf-8") + mm.load_control_inputs(protocol, roster, mm.control_text_sha(protocol)) + + +def test_tracked_control_files_have_no_machine_paths(monkeypatch): + mm = _import_cli_module(monkeypatch) + framework_marker = "AgentBench" + "Framework.framework" + markers = ("C:" + "/Users/", "C:" + "\\\\Users\\\\", "/home/", "/Users/", framework_marker) + for path in (mm.PROTOCOL, mm.ROSTER, mm.REPO / "vendor" / "miracle_local" / "run_match.py"): + text = path.read_text(encoding="utf-8") + for marker in markers: + assert marker not in text + + +def test_load_control_inputs_rejects_non_contiguous_roster(tmp_path, monkeypatch): + mm = _import_cli_module(monkeypatch) + protocol_path = tmp_path / "protocol.json" + roster_path = tmp_path / "roster.json" + protocol_path.write_text("{}", encoding="utf-8") + roster_path.write_text( + json.dumps({"strategies": [{"rank": 1}, {"rank": 3}]}), + encoding="utf-8", + ) + + with pytest.raises(mm.PreflightError, match="roster ranks"): + mm.load_control_inputs(protocol_path, roster_path) + + +def test_load_control_inputs_rejects_non_object_strategy(tmp_path, monkeypatch): + mm = _import_cli_module(monkeypatch) + protocol_path = tmp_path / "protocol.json" + roster_path = tmp_path / "roster.json" + protocol_path.write_text("{}", encoding="utf-8") + roster_path.write_text( + json.dumps([{"rank": rank} for rank in range(1, 17)] + ["not-an-object"]), + encoding="utf-8", + ) + + with pytest.raises(mm.PreflightError, match="roster JSON must be an object"): + mm.load_control_inputs(protocol_path, roster_path) + + +def test_cli_module_import_does_not_require_external_env(monkeypatch): + repo = Path(__file__).resolve().parents[2] + monkeypatch.delenv("AGENTBENCH_ROOT", raising=False) + monkeypatch.delenv("MIRACLE_IFELSE_DIR", raising=False) + sys.modules.pop("miracle_matrix", None) + tools_dir = str(repo / "tools") + if tools_dir not in sys.path: + sys.path.insert(0, tools_dir) + + mm = importlib.import_module("miracle_matrix") + + assert mm.PROTOCOL.name == "24_miracle_evaluation_protocol.v0.3.json" diff --git a/tests/miracle/test_matrix_identity.py b/tests/miracle/test_matrix_identity.py new file mode 100644 index 0000000..2005e4d --- /dev/null +++ b/tests/miracle/test_matrix_identity.py @@ -0,0 +1,121 @@ +from __future__ import annotations + +import hashlib +import importlib +import sys +from pathlib import Path + +import pytest + + +def _import_cli_module(monkeypatch): + repo = Path(__file__).resolve().parents[2] + tools_dir = str(repo / "tools") + if tools_dir not in sys.path: + sys.path.insert(0, tools_dir) + monkeypatch.setenv("AGENTBENCH_ROOT", str(repo)) + monkeypatch.setenv("MIRACLE_IFELSE_DIR", str(repo)) + sys.modules.pop("miracle_matrix", None) + return importlib.import_module("miracle_matrix") + + +def _sha(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def test_resolve_unique_dir_rejects_ambiguous_matches(tmp_path, monkeypatch): + mm = _import_cli_module(monkeypatch) + (tmp_path / "rank04__a").mkdir() + (tmp_path / "rank04__b").mkdir() + + with pytest.raises(RuntimeError, match="multiple opponent directories"): + mm.resolve_unique_dir(tmp_path, "rank04__*") + + +def test_verify_python_hashes_detects_modified_entry(tmp_path, monkeypatch): + mm = _import_cli_module(monkeypatch) + python_dir = tmp_path / "rank04__fixture" + python_dir.mkdir() + (python_dir / "main.py").write_text("modified\n", encoding="utf-8") + strategy = { + "rank": 4, + "type": "python_script", + "entry": "main.py", + "runnable_sha256": "expected", + } + + errors = mm.verify_python_strategy_hashes(strategy, tmp_path) + + assert any("rank04" in error and "runnable sha" in error for error in errors) + + +def test_verify_python_hashes_checks_entry(tmp_path, monkeypatch): + mm = _import_cli_module(monkeypatch) + extracted = tmp_path / "extracted" + archives = tmp_path / "archives" + extracted.mkdir() + archives.mkdir() + python_dir = extracted / "rank04__fixture" + python_dir.mkdir() + entry = python_dir / "main.py" + entry.write_text("print('ok')\n", encoding="utf-8") + archive = archives / "rank04__fixture.zip" + archive.write_bytes(b"archive-bytes") + strategy = { + "rank": 4, + "type": "python_script", + "entry": "main.py", + "runnable_sha256": _sha(entry), + "archive_sha256": _sha(archive), + } + + errors = mm.verify_python_strategy_hashes(strategy, extracted) + + assert errors == [] + + +def test_verify_archive_hash_rejects_duplicate_archives(tmp_path, monkeypatch): + mm = _import_cli_module(monkeypatch) + extracted = tmp_path / "extracted" + archives = tmp_path / "archives" + extracted.mkdir() + archives.mkdir() + python_dir = extracted / "rank04__fixture" + python_dir.mkdir() + (python_dir / "main.py").write_text("print('ok')\n", encoding="utf-8") + (archives / "rank04__a.zip").write_bytes(b"a") + (archives / "rank04__b.zip").write_bytes(b"b") + strategy = { + "rank": 4, + "type": "python_script", + "entry": "main.py", + "runnable_sha256": _sha(python_dir / "main.py"), + "archive_sha256": _sha(archives / "rank04__a.zip"), + } + + errors = mm.verify_archive_hash(strategy, archives) + + assert any("multiple archive files" in error for error in errors) + + +def test_verify_archive_hash_requires_archive_identity(tmp_path, monkeypatch): + mm = _import_cli_module(monkeypatch) + extracted = tmp_path / "extracted" + archives = tmp_path / "archives" + extracted.mkdir() + archives.mkdir() + python_dir = extracted / "rank04__fixture" + python_dir.mkdir() + entry = python_dir / "main.py" + entry.write_text("print('ok')\n", encoding="utf-8") + (archives / "rank04__fixture.zip").write_bytes(b"archive-bytes") + strategy = { + "rank": 4, + "type": "python_script", + "entry": "main.py", + "runnable_sha256": _sha(entry), + } + + errors = mm.verify_archive_hash(strategy, archives) + + assert any("archive sha missing" in error for error in errors) diff --git a/tests/miracle/test_matrix_resume_assets.py b/tests/miracle/test_matrix_resume_assets.py new file mode 100644 index 0000000..4bfa382 --- /dev/null +++ b/tests/miracle/test_matrix_resume_assets.py @@ -0,0 +1,176 @@ +from __future__ import annotations + +import importlib +import json +import sys +from pathlib import Path + +import pytest + + +PYTHON_RANKS = {4, 5, 7} + + +def _import_cli_module(monkeypatch): + repo = Path(__file__).resolve().parents[2] + tools_dir = str(repo / "tools") + if tools_dir not in sys.path: + sys.path.insert(0, tools_dir) + monkeypatch.setenv("AGENTBENCH_ROOT", str(repo)) + monkeypatch.setenv("MIRACLE_IFELSE_DIR", str(repo)) + sys.modules.pop("miracle_matrix", None) + return importlib.import_module("miracle_matrix") + + +def _tree_bytes(root: Path) -> dict[str, bytes]: + return { + str(path.relative_to(root)): path.read_bytes() + for path in sorted(root.rglob("*")) if path.is_file() + } + + +def _make_assets(tmp_path: Path, mm): + roots = {name: tmp_path / name for name in ( + "judge", "ifelse", "extracted", "archives", "precheck", "rank16", + )} + for root in roots.values(): + root.mkdir() + (roots["judge"] / "main.py").write_text("judge\n", encoding="utf-8") + (roots["ifelse"] / "main.py").write_text("ifelse\n", encoding="utf-8") + + strategies = [] + build_hashes = {} + archive_paths = {} + executable_paths = {} + for rank in range(1, 17): + archive = roots["archives"] / f"rank{rank:02d}__fixture.zip" + archive.write_bytes(f"archive-{rank}".encode("ascii")) + archive_paths[rank] = archive + strategy = {"rank": rank, "archive_sha256": mm.sha(archive)} + if rank in PYTHON_RANKS: + directory = roots["extracted"] / f"rank{rank:02d}__fixture" + directory.mkdir() + entry = directory / "main.py" + entry.write_text(f"python-{rank}\n", encoding="utf-8") + strategy.update({"entry": "main.py", "runnable_sha256": mm.sha(entry)}) + executable_paths[rank] = entry + else: + if rank == 16: + directory = roots["rank16"] / "rank16_copy" + else: + directory = roots["precheck"] / "strategies" / f"rank{rank:02d}" + directory.mkdir(parents=True) + executable = directory / "main.exe" + executable.write_bytes(f"exe-{rank}".encode("ascii")) + executable_paths[rank] = executable + build_hashes[f"rank{rank:02d}"] = mm.sha(executable) + strategies.append(strategy) + + protocol = { + "frozen_identities": { + "build_artifacts_win64_mingw": build_hashes, + "evaluated_agent": {"sha256": mm.sha(roots["ifelse"] / "main.py")}, + "judge": {"main_py_sha256": mm.sha(roots["judge"] / "main.py")}, + } + } + control = mm.ControlInputs(protocol, {"strategies": strategies}, {"protocol": "p", "roster": "r"}) + return roots, control, archive_paths, executable_paths + + +class _FakeRunner: + def __init__(self, session_dir: Path): + self.session_dir = session_dir + self.resume_count = 0 + self.execute_count = 0 + + def resume(self, _sid): + self.resume_count += 1 + + def execute(self): + self.execute_count += 1 + return {"completed": True, "halted": False} + + def write_run_compatible_output(self): + return self.session_dir + + def aggregate_from_events(self): + return {"total_attempts": 0, "valid_games": 0, "invalid_games": 0, "win_rate": None} + + +def _run_with_assets(tmp_path, monkeypatch, mutate=None, missing_root=None): + mm = _import_cli_module(monkeypatch) + roots, control, archives, executables = _make_assets(tmp_path, mm) + session = tmp_path / "session"; session.mkdir() + (session / "manifest.json").write_text(json.dumps({"fixture": True}), encoding="utf-8") + (session / "progress.json").write_text('{"attempts": {}}', encoding="utf-8") + monkeypatch.setattr(mm, "verify_session_for_resume", lambda *_a, **_k: (True, [])) + if mutate is not None: + mutate(roots, archives, executables) + runner = _FakeRunner(session) + before = _tree_bytes(session) + kwargs = { + "judge_dir": roots["judge"], "ifelse_dir": roots["ifelse"], + "extracted_root": roots["extracted"], "archives_root": roots["archives"], + "precheck_root": roots["precheck"], "rank16_build_root": roots["rank16"], + } + if missing_root is not None: + kwargs[missing_root] = None + rc = mm._run_resume(runner, "session", control_inputs=control, session_root=tmp_path, **kwargs) + return rc, runner, session, before + + +@pytest.mark.parametrize("name,mutate", [ + ("judge", lambda roots, _archives, _executables: (roots["judge"] / "main.py").write_text("changed\n", encoding="utf-8")), + ("ifelse", lambda roots, _archives, _executables: (roots["ifelse"] / "main.py").write_text("changed\n", encoding="utf-8")), + ("python runnable", lambda _roots, _archives, executables: executables[4].write_text("changed\n", encoding="utf-8")), + ("python archive", lambda _roots, archives, _executables: archives[4].write_bytes(b"changed")), + ("c++ archive", lambda _roots, archives, _executables: archives[1].write_bytes(b"changed")), + ("c++ executable", lambda _roots, _archives, executables: executables[1].write_bytes(b"changed")), + ("c++ executable is directory", lambda _roots, _archives, executables: (executables[1].unlink(), executables[1].mkdir())), + ("archive missing", lambda _roots, archives, _executables: archives[1].unlink()), + ("archive duplicate", lambda roots, _archives, _executables: (roots["archives"] / "rank01__duplicate.zip").write_bytes(b"duplicate")), +]) +def test_resume_rejects_changed_real_runtime_assets_before_session_write(tmp_path, monkeypatch, name, mutate): + rc, runner, session, before = _run_with_assets(tmp_path, monkeypatch, mutate=mutate) + + assert rc == 2, name + assert runner.resume_count == 0 + assert runner.execute_count == 0 + assert _tree_bytes(session) == before + assert not (session / "matrix.full.log").exists() + + +@pytest.mark.parametrize("rank", range(1, 17)) +def test_resume_verifies_the_unique_archive_for_every_rank(tmp_path, monkeypatch, rank): + rc, runner, session, before = _run_with_assets( + tmp_path, monkeypatch, + mutate=lambda _roots, archives, _executables: archives[rank].write_bytes(b"changed"), + ) + + assert rc == 2 + assert runner.resume_count == 0 + assert runner.execute_count == 0 + assert _tree_bytes(session) == before + assert not (session / "matrix.full.log").exists() + + +@pytest.mark.parametrize("missing_root", [ + "judge_dir", "ifelse_dir", "extracted_root", "archives_root", "precheck_root", "rank16_build_root", +]) +def test_resume_rejects_any_missing_runtime_root_before_session_write(tmp_path, monkeypatch, capsys, missing_root): + rc, runner, session, before = _run_with_assets(tmp_path, monkeypatch, missing_root=missing_root) + + assert rc == 2 + assert f"runtime root missing: {missing_root}" in capsys.readouterr().err + assert runner.resume_count == 0 + assert runner.execute_count == 0 + assert _tree_bytes(session) == before + assert not (session / "matrix.full.log").exists() + + +def test_resume_with_all_real_runtime_assets_calls_resume_once(tmp_path, monkeypatch): + rc, runner, _session, _before = _run_with_assets(tmp_path, monkeypatch) + + assert rc == 0 + assert runner.resume_count == 1 + assert runner.execute_count == 1 diff --git a/tests/miracle/test_matrix_runner.py b/tests/miracle/test_matrix_runner.py new file mode 100644 index 0000000..df512ea --- /dev/null +++ b/tests/miracle/test_matrix_runner.py @@ -0,0 +1,218 @@ +"""Tests for the matrix runner (tools/miracle_matrix.py logic). + +Uses an injected FAKE attempt_fn (signature matches the runner's keyword call: +``attempt_fn(game_id=..., p0_dir=..., ...)``) — no Judge, no AI subprocess. +Covers the 15 runner safety requirements. +""" +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from agentbench_frame.games.miracle import matrix_runner +from agentbench_frame.games.miracle.matrix_runner import MatrixRunner + + +def _att(game_id, **over): + rc = matrix_runner.parse_game_id(game_id) + d = dict(valid=True, normalized_result="win", raw_winner=None, error_type=None, + wrapper_timeout=False, result_json_status="ok", discrepancies=None, + realized_randomization=None, scores=None, steps=10, ai_crash_player=None, + judge_crash=False) + d.update(over) + rank, camp = rc + rw = d["raw_winner"] if d["raw_winner"] is not None else ( + camp if d["normalized_result"] == "win" else 1 - camp) + class A: pass + a = A() + a.game_id = game_id; a.rank = rank; a.camp = camp + a.valid = d["valid"]; a.normalized_result = d["normalized_result"]; a.raw_winner = rw + a.error_type = d["error_type"]; a.wrapper_timeout = d["wrapper_timeout"] + a.result_json_status = d["result_json_status"]; a.discrepancies = d["discrepancies"] or [] + a.realized_randomization = d["realized_randomization"] or {"map_type": 0, "day_time": 1} + a.scores = d["scores"] or {"0": 5, "1": 2}; a.steps = d["steps"] + a.ai_crash_player = d["ai_crash_player"]; a.judge_crash = d["judge_crash"] + a.winner_agent = "miracle_ifelse" if d["normalized_result"] == "win" else "opp" + a.normal_cleanup_nonzero = False; a.reason = d["error_type"] or "" + a.evidence_paths = {"stdout": "", "stderr": "", "trace": "", "replay": "", + "result_json": str(Path(game_id + ".rj"))} + return a + + +def _runner(tmp_path, fn, **kw): + return MatrixRunner(session_root=tmp_path, judge_dir=tmp_path / "j", + ifelse_dir=tmp_path / "ifelse", opponent_dir_of=lambda r: tmp_path / f"opp{r}", + vendor_script=tmp_path / "v.py", framework_src=tmp_path / "src", + timeout=8.0, wrapper_timeout_s=60.0, attempt_fn=fn, + protocol_sha="p" * 64, run_id="RID", **kw) + + +# 1. dry-run never starts Judge/AI +def test_dry_run_starts_nothing(tmp_path): + called = [] + r = _runner(tmp_path, lambda **k: called.append(1)) + r.prepare_session() + out = r.dry_run() + assert called == [] + assert out["plan_count"] == 32 + + +# 2. plan rank01-16 camp0/camp1, 32 unique +def test_dry_run_plan_correct(tmp_path): + r = _runner(tmp_path, lambda **k: None) + r.prepare_session() + out = r.dry_run() + ids = [a["game_id"] for a in out["plan"]] + assert len(ids) == 32 and len(set(ids)) == 32 + assert out["plan"][0]["game_id"] == "m_rank01_camp0" + assert out["plan"][-1]["game_id"] == "m_rank16_camp1" + + +# 3. existing session refused +def test_existing_session_refused(tmp_path): + r = _runner(tmp_path, lambda **k: None) + r.prepare_session() + r2 = _runner(tmp_path, lambda **k: None) + with pytest.raises(FileExistsError): + r2.prepare_session_for_existing(r.session_id) + + +# 4+5. manifest atomic + hashes +def test_manifest_atomic_with_hashes(tmp_path): + r = _runner(tmp_path, lambda **k: None) + r.prepare_session() + r.record_manifest(opponent_hashes={i: "h" + str(i) for i in range(1, 17)}, + build_hashes={i: "b" + str(i) for i in [1, 2, 3, 6] + list(range(8, 17))}, + ifelse_sha="IF", judge_sha="JD", code_hashes={"matrix_runner": "MR"}) + m = json.loads((r.session_dir / "manifest.json").read_text(encoding="utf-8")) + assert m["protocol_sha256"] == "p" * 64 + assert m["plan_count"] == 32 and m["run_id"] == "RID" + assert m["ifelse_sha256"] == "IF" and m["judge_sha256"] == "JD" + assert not (r.session_dir / "manifest.json.tmp").exists() + + +def test_manifest_records_control_input_hashes(tmp_path): + r = _runner(tmp_path, lambda **k: None) + r.prepare_session() + r.record_manifest( + opponent_hashes={i: "h" + str(i) for i in range(1, 17)}, + build_hashes={i: "b" + str(i) for i in range(1, 17)}, + ifelse_sha="IF", judge_sha="JD", code_hashes={}, + control_inputs={"protocol": {"path": "/p.json", "sha256": "p"}, + "roster": {"path": "/r.json", "sha256": "r"}}, + ) + manifest = json.loads((r.session_dir / "manifest.json").read_text(encoding="utf-8")) + assert manifest["control_inputs"]["protocol"]["sha256"] == "p" + + +# 6. running -> UNCERTAIN_IN_FLIGHT, no auto-rerun +def test_running_state_uncertain_no_rerun(tmp_path): + r = _runner(tmp_path, lambda **k: None) + r.prepare_session() + plan = matrix_runner.make_attempt_plan() + matrix_runner.mark_running(r.progress, plan[0]["game_id"]) + matrix_runner.write_progress_atomic(r.progress_path, r.progress) + res = r.execute() + assert res["halted"] is True + assert "UNCERTAIN_IN_FLIGHT" in res["reason"] + + +# 7. done only after attempt returns +def test_done_only_after_attempt_returns(tmp_path): + seen = [] + def fn(**k): + seen.append(k["game_id"]); return _att(k["game_id"], normalized_result="win") + r = _runner(tmp_path, fn) + r.prepare_session() + r.execute_rank(rank=1) + prog = json.loads(r.progress_path.read_text(encoding="utf-8")) + assert prog["attempts"]["m_rank01_camp0"]["state"] == "done" + assert prog["attempts"]["m_rank01_camp1"]["state"] == "done" + + +# 8+9. valid/AI-invalid/infra distinct; AI-invalid continues, not in win-rate denom +def test_ai_invalid_continues_not_in_winrate(tmp_path): + calls = [0] + def fn(**k): + i = calls[0]; calls[0] += 1 + if i == 0: + return _att(k["game_id"], normalized_result="error", error_type="ai_crash", valid=False) + return _att(k["game_id"], normalized_result="win") + r = _runner(tmp_path, fn) + r.prepare_session() + r.execute_rank(rank=1) + agg = r.aggregate_from_events() + assert agg["total_attempts"] == 2 and agg["valid_games"] == 1 and agg["invalid_games"] == 1 + assert agg["wins"] == 1 and agg["win_rate"] == pytest.approx(1.0) + + +# 10. infra failure blocks next +def test_infra_failure_halts(tmp_path): + calls = [0] + def fn(**k): + calls[0] += 1 + if calls[0] == 1: + return _att(k["game_id"], error_type="evidence_mismatch", valid=False) + return _att(k["game_id"]) + r = _runner(tmp_path, fn) + r.prepare_session() + res = r.execute_rank(rank=1) + assert res["halted"] is True and res["state"] == "HALTED_INFRA_FAILURE" + assert calls[0] == 1 + + +# 11. rank audit after two games +def test_rank_audit_runs_after_two_games(tmp_path): + r = _runner(tmp_path, lambda **k: _att(k["game_id"])) + r.prepare_session() + r.execute_rank(rank=1) + assert (r.session_dir / "audit" / "rank01.json").exists() + + +# 12. progress/events no duplicate +def test_progress_and_events_no_duplicate(tmp_path): + r = _runner(tmp_path, lambda **k: _att(k["game_id"])) + r.prepare_session() + r.execute_rank(rank=1) + ev = [json.loads(l) for l in r.events_path.read_text(encoding="utf-8").splitlines() if l.strip()] + ids = [e["game_id"] for e in ev] + assert len(ids) == len(set(ids)) + + +# 13. normal cleanup nonzero not crash +def test_normal_cleanup_nonzero_not_crash(tmp_path): + def fn(**k): + a = _att(k["game_id"]); a.normal_cleanup_nonzero = True; return a + r = _runner(tmp_path, fn) + r.prepare_session() + res = r.execute_rank(rank=1) + assert res["halted"] is False + + +# 14. PID residual blocks continue +def test_residual_blocks_continue(tmp_path): + import os as _os, psutil + me = psutil.Process(_os.getpid()) + def fn(**k): + a = _att(k["game_id"]) + rjp = r.session_dir / (k["game_id"] + ".rj") + rjp.write_text(json.dumps({"judge": {"pid": _os.getpid(), "started_at": me.create_time(), "role": "judge"}})) + a.evidence_paths = {"result_json": str(rjp)} + return a + r = _runner(tmp_path, fn) + r.prepare_session() + res = r.execute_rank(rank=1) + assert res["halted"] is True and "residual" in res["reason"] + + +# 15. partial/complete summary independently re-computable +def test_summary_recomputable_from_events(tmp_path): + r = _runner(tmp_path, lambda **k: _att(k["game_id"])) + r.prepare_session() + r.execute_rank(rank=1) + agg = r.aggregate_from_events() + raw = [json.loads(l) for l in r.events_path.read_text(encoding="utf-8").splitlines() if l.strip()] + assert agg["total_attempts"] == 2 and len(raw) == 2 + assert agg["wins"] == sum(1 for e in raw if e["normalized_result"] == "win") diff --git a/tests/miracle/test_section2_strictness.py b/tests/miracle/test_section2_strictness.py new file mode 100644 index 0000000..f0ec3b3 --- /dev/null +++ b/tests/miracle/test_section2_strictness.py @@ -0,0 +1,620 @@ +"""Section 2 余下严格性缺口针对性红灯测试(24_miracle resume 验证链)。 + +覆盖上一位 Agent 完成的 verify_session_for_resume 与 resume() 之外的剩余缺口: + + 1. progress 中未知 game_id 必须明确拒绝 + 2. progress 中非法 state(非 not_started/running/done)必须明确拒绝 + 3. opponent / build / code_hashes 字典必须精确键集合匹配(拒绝额外键) + 4. manifest 缺 session_id 必须拒绝 + 5. plan_count 必须与真实 plan 长度 AND 32 严格一致 + 6. complete-rank audit 必须解析、rank 一致、ok=True、两 game_id/camp 与 progress 一致 + 7. partial rank(仅一 camp done)提前出现声称完整的 audit 文件必须拒绝 + 8. CLI 级 fake 恢复必须经过 tools/miracle_matrix._run_resume 的 verify 链 + 9. resume() 后不应再走宽松 load_progress(必须严格解析同一文件) + 10. events.jsonl 非 JSON 对象行(int/str/list 等顶层非 dict)必须保守拒绝 + +只读:所有测试均使用 tmp_path 临时夹具,不触碰权威 session;不启动 Judge/AI; +无 git 操作。 +""" +from __future__ import annotations + +import json +import os +import sys +from pathlib import Path + +import pytest + +from agentbench_frame.games.miracle.matrix_runner import verify_session_for_resume + + +# --------------------------------------------------------------------------- # +# helpers +# --------------------------------------------------------------------------- # +_PLAN32 = [{"rank": r, "camp": c, "game_id": f"m_rank{r:02d}_camp{c}"} + for r in range(1, 17) for c in (0, 1)] +_KNOWN_GIDS = {a["game_id"] for a in _PLAN32} + + +def _wm(tmp_path, *, progress=None, **over): + """Write a full manifest with all identity fields. session_id 默认等于 dirname.""" + m = { + "protocol_sha256": "a" * 64, + "code_hashes": {"matrix": "abc", "matrix_runner": "def", + "match_runner": "ghi", "vendor_run_match": "jkl"}, + "plan_count": 32, "plan": _PLAN32, "run_id": "RID", + "timeout": 8.0, "wrapper_timeout_s": 180.0, + "ifelse_sha256": "if_sha", "judge_sha256": "j_sha", + "opponent_archive_sha256": {f"rank{r:02d}": f"opp_{r}" for r in range(1, 17)}, + "cpp_build_sha256": {f"rank{r:02d}": f"bld_{r}" for r in range(1, 17)}, + "session_id": tmp_path.name, "python_version": "3.13.5", "platform": "test_plat", + } + m.update(over) + (tmp_path / "manifest.json").write_text(json.dumps(m, ensure_ascii=False), encoding="utf-8") + pr = progress if progress is not None else {"attempts": {}} + (tmp_path / "progress.json").write_text(json.dumps(pr, ensure_ascii=False), encoding="utf-8") + return m + + +def _done_entry(*, valid=True, norm="win", rank=1, camp=0): + return {"state": "done", "valid": valid, "normalized_result": norm, + "rank": rank, "camp": camp} + + +# =========================================================================== # +# 1. progress 未知 game_id 拒绝 +# =========================================================================== # +def test_progress_unknown_game_id_rejected(tmp_path): + proc = {"attempts": {"m_rank01_camp0": _done_entry(rank=1, camp=0), + "m_rank99_camp5": _done_entry(rank=99, camp=5)}} + _wm(tmp_path, progress=proc) + ok, errs = verify_session_for_resume(tmp_path) + assert not ok + assert any("unknown" in e.lower() or "m_rank99" in e for e in errs), errs + + +# =========================================================================== # +# 2. progress 非法 state 拒绝 +# =========================================================================== # +def test_progress_illegal_state_rejected(tmp_path): + proc = {"attempts": {"m_rank01_camp0": {"state": "tilted", "rank": 1, "camp": 0}}} + _wm(tmp_path, progress=proc) + ok, errs = verify_session_for_resume(tmp_path) + assert not ok + assert any("state" in e.lower() and ("illegal" in e.lower() or "invalid" in e.lower()) + for e in errs), errs + + +def test_progress_not_started_explicit_state_accepted(tmp_path): + """state=not_started 显式出现也应被接受(合法,无事件对应)。""" + proc = {"attempts": {"m_rank01_camp0": {"state": "not_started", "rank": 1, "camp": 0}}} + _wm(tmp_path, progress=proc) + # without events / running it should be all clean — but we don't pass expected_* IDs + # so the unknown check must NOT flag a known id with legal not_started state. + ok, errs = verify_session_for_resume(tmp_path) + assert not any("m_rank01_camp0" in e and "unknown" in e.lower() for e in errs), errs + + +# =========================================================================== # +# 3. opponent/build/code_hashes 精确键集合匹配(拒绝额外键) +# =========================================================================== # +def test_manifest_opponent_extra_key_rejected(tmp_path): + m = _wm(tmp_path) + m["opponent_archive_sha256"]["rank99"] = "should_not_be_here" + (tmp_path / "manifest.json").write_text(json.dumps(m, ensure_ascii=False), encoding="utf-8") + ok, errs = verify_session_for_resume( + tmp_path, + expected_opponent_shas={r: f"opp_{r}" for r in range(1, 17)}, + ) + assert not ok + assert any("opponent" in e.lower() and ("extra" in e.lower() or "unexpected" in e.lower()) + for e in errs), errs + + +def test_manifest_build_extra_key_rejected(tmp_path): + m = _wm(tmp_path) + m["cpp_build_sha256"]["rank99"] = "should_not_be_here" + (tmp_path / "manifest.json").write_text(json.dumps(m, ensure_ascii=False), encoding="utf-8") + ok, errs = verify_session_for_resume( + tmp_path, + expected_build_shas={r: f"bld_{r}" for r in range(1, 17)}, + ) + assert not ok + assert any("build" in e.lower() and ("extra" in e.lower() or "unexpected" in e.lower()) + for e in errs), errs + + +def test_manifest_code_hash_extra_key_rejected(tmp_path): + m = _wm(tmp_path) + m["code_hashes"]["rogue_module"] = "should_not_be_here" + (tmp_path / "manifest.json").write_text(json.dumps(m, ensure_ascii=False), encoding="utf-8") + ok, errs = verify_session_for_resume( + tmp_path, + code_files={"matrix": str(tmp_path / "manifest.json"), # any existing file + "matrix_runner": str(tmp_path / "manifest.json"), + "match_runner": str(tmp_path / "manifest.json")}, + ) + assert not ok + assert any("code" in e.lower() and ("extra" in e.lower() or "unexpected" in e.lower()) + for e in errs), errs + + +def test_resume_rejects_control_input_hash_change(tmp_path): + m = _wm(tmp_path) + m["control_inputs"] = { + "protocol": {"path": "/p.json", "sha256": "old"}, + "roster": {"path": "/r.json", "sha256": "same"}, + } + (tmp_path / "manifest.json").write_text(json.dumps(m, ensure_ascii=False), encoding="utf-8") + ok, errors = verify_session_for_resume( + tmp_path, + expected_control_inputs={"protocol": "new", "roster": "same"}, + ) + assert not ok + assert any("control input hash mismatch" in error for error in errors), errors + + +def test_manifest_opponent_missing_key_rejected(tmp_path): + m = _wm(tmp_path) + del m["opponent_archive_sha256"]["rank16"] + (tmp_path / "manifest.json").write_text(json.dumps(m, ensure_ascii=False), encoding="utf-8") + ok, errs = verify_session_for_resume( + tmp_path, + expected_opponent_shas={r: f"opp_{r}" for r in range(1, 17)}, + ) + assert not ok + assert any("opponent" in e.lower() for e in errs), errs + + +# =========================================================================== # +# 4. manifest 缺 session_id 必须拒绝 +# =========================================================================== # +def test_manifest_missing_session_id_rejected(tmp_path): + m = _wm(tmp_path) + del m["session_id"] + (tmp_path / "manifest.json").write_text(json.dumps(m, ensure_ascii=False), encoding="utf-8") + ok, errs = verify_session_for_resume(tmp_path) + assert not ok + assert any("session_id" in e.lower() and ("missing" in e.lower() or "absent" in e.lower()) + for e in errs), errs + + +# =========================================================================== # +# 5. plan_count 与 plan 长度 AND 32 严格一致 +# =========================================================================== # +def test_manifest_plan_count_mismatch_len_rejected(tmp_path): + """plan_count 写 5 但 plan 还是 32 项 -> 拒绝。""" + _wm(tmp_path, plan_count=5) + ok, errs = verify_session_for_resume(tmp_path) + assert not ok + assert any("plan_count" in e.lower() for e in errs), errs + + +def test_manifest_plan_count_field_missing_rejected(tmp_path): + m = _wm(tmp_path) + del m["plan_count"] + (tmp_path / "manifest.json").write_text(json.dumps(m, ensure_ascii=False), encoding="utf-8") + ok, errs = verify_session_for_resume(tmp_path) + assert not ok + assert any("plan_count" in e.lower() for e in errs), errs + + +def test_manifest_plan_count_correct_passes(tmp_path): + """plan_count=32 且 plan 长 32 时不报错(正反向安抚)。""" + _wm(tmp_path, plan_count=32) + ok, errs = verify_session_for_resume(tmp_path) + assert not any("plan_count" in e.lower() for e in errs), errs + + +# =========================================================================== # +# 6. complete-rank audit 必须解析、rank 一致、ok=True、game_id/camp 与 progress 一致 +# =========================================================================== # +def _mk_audit_dir(tmp_path, rank): + d = tmp_path / "audit" + d.mkdir(exist_ok=True) + return d / f"rank{rank:02d}.json" + + +def test_complete_rank_audit_corrupt_rejected(tmp_path): + proc = {"attempts": { + "m_rank01_camp0": _done_entry(rank=1, camp=0), + "m_rank01_camp1": _done_entry(rank=1, camp=1, norm="loss"), + }} + _wm(tmp_path, progress=proc) + af = _mk_audit_dir(tmp_path, 1) + af.write_text("NOT JSON {{{", encoding="utf-8") + ok, errs = verify_session_for_resume(tmp_path) + assert not ok + assert any("rank01" in e and ("audit" in e.lower() and "corrupt" in e.lower()) + for e in errs), errs + + +def test_complete_rank_audit_rank_wrong_rejected(tmp_path): + proc = {"attempts": { + "m_rank01_camp0": _done_entry(rank=1, camp=0), + "m_rank01_camp1": _done_entry(rank=1, camp=1, norm="loss"), + }} + _wm(tmp_path, progress=proc) + af = _mk_audit_dir(tmp_path, 1) + af.write_text(json.dumps({"rank": 99, "ok": True, "games": []}), encoding="utf-8") + ok, errs = verify_session_for_resume(tmp_path) + assert not ok + assert any("rank01" in e and "audit" in e.lower() for e in errs), errs + + +def test_complete_rank_audit_not_ok_rejected(tmp_path): + proc = {"attempts": { + "m_rank01_camp0": _done_entry(rank=1, camp=0), + "m_rank01_camp1": _done_entry(rank=1, camp=1, norm="loss"), + }} + _wm(tmp_path, progress=proc) + af = _mk_audit_dir(tmp_path, 1) + af.write_text(json.dumps({"rank": 1, "ok": False, "reasons": ["x"], + "games": [{"game_id": "m_rank01_camp0", "camp": 0}, + {"game_id": "m_rank01_camp1", "camp": 1}]}), + encoding="utf-8") + ok, errs = verify_session_for_resume(tmp_path) + assert not ok + assert any("rank01" in e and "audit" in e.lower() and "ok" in e.lower() for e in errs), errs + + +def test_complete_rank_audit_game_ids_mismatch_rejected(tmp_path): + """audit JSON 声称的 games 与 progress 的 done 状态 game_id 不一致 -> 拒绝。""" + proc = {"attempts": { + "m_rank01_camp0": _done_entry(rank=1, camp=0), + "m_rank01_camp1": _done_entry(rank=1, camp=1, norm="loss"), + }} + _wm(tmp_path, progress=proc) + af = _mk_audit_dir(tmp_path, 1) + af.write_text(json.dumps({"rank": 1, "ok": True, + "games": [{"game_id": "m_rank01_camp0", "camp": 0}, + {"game_id": "m_rank01_camp0", "camp": 0}]}), + encoding="utf-8") + ok, errs = verify_session_for_resume(tmp_path) + assert not ok + assert any("rank01" in e and "audit" in e.lower() + and ("game_id" in e.lower() or "mismatch" in e.lower()) for e in errs), errs + + +def test_complete_rank_audit_camps_mismatch_rejected(tmp_path): + proc = {"attempts": { + "m_rank01_camp0": _done_entry(rank=1, camp=0), + "m_rank01_camp1": _done_entry(rank=1, camp=1, norm="loss"), + }} + _wm(tmp_path, progress=proc) + af = _mk_audit_dir(tmp_path, 1) + af.write_text(json.dumps({"rank": 1, "ok": True, + "games": [{"game_id": "m_rank01_camp0", "camp": 1}, + {"game_id": "m_rank01_camp1", "camp": 0}]}), + encoding="utf-8") + ok, errs = verify_session_for_resume(tmp_path) + assert not ok + assert any("rank01" in e and "audit" in e.lower() for e in errs), errs + + +def test_complete_rank_audit_valid_passes(tmp_path): + proc = {"attempts": { + "m_rank01_camp0": _done_entry(rank=1, camp=0), + "m_rank01_camp1": _done_entry(rank=1, camp=1, norm="loss"), + }} + _wm(tmp_path, progress=proc) + af = _mk_audit_dir(tmp_path, 1) + af.write_text(json.dumps({"rank": 1, "ok": True, "reasons": [], + "games": [{"game_id": "m_rank01_camp0", "camp": 0}, + {"game_id": "m_rank01_camp1", "camp": 1}]}), + encoding="utf-8") + ok, errs = verify_session_for_resume(tmp_path) + # the only errs possibly left must NOT be about audit + assert not any("rank01" in e and "audit" in e.lower() for e in errs), errs + + +# =========================================================================== # +# 7. partial rank 提前出现声称完整的 audit 文件必须拒绝 +# =========================================================================== # +def test_partial_rank_premature_complete_audit_rejected(tmp_path): + """只有 camp0 done(part),却已存在一份标准 audit —— 必须拒绝。""" + proc = {"attempts": { + "m_rank01_camp0": _done_entry(rank=1, camp=0), + # camp1 not_started (absent from attempts) + }} + _wm(tmp_path, progress=proc) + af = _mk_audit_dir(tmp_path, 1) + af.write_text(json.dumps({"rank": 1, "ok": True, "reasons": [], + "games": [{"game_id": "m_rank01_camp0", "camp": 0}, + {"game_id": "m_rank01_camp1", "camp": 1}]}), + encoding="utf-8") + ok, errs = verify_session_for_resume(tmp_path) + assert not ok + assert any("rank01" in e.lower() and + ("premature" in e.lower() or "partial" in e.lower() or "pre" in e.lower()) + for e in errs), errs + + +# =========================================================================== # +# 8. CLI 级 fake 恢复必须经过 tools/miracle_matrix._run_resume 的 verify 链 +# =========================================================================== # +def _import_cli_module(monkeypatch): + """Import tools/miracle_matrix as a module. The module top-level calls + paths.judge_dir() / ifelse_dir() which require AGENTBENCH_ROOT and + MIRACLE_IFELSE_DIR env vars to be set (they only BUILD paths, never read + them, so dummy values are safe — no Judge/AI is touched).""" + repo = Path(__file__).resolve().parents[2] + tools_dir = str(repo / "tools") + if tools_dir not in sys.path: + sys.path.insert(0, tools_dir) + # tools/miracle_matrix.py only ever uses these to construct Path objects; + # it never reads the file system at import time. Dummy values are safe and + # do NOT touch the authoritative Judge/AI matrix (see SKILL.md boundary). + monkeypatch.setenv("AGENTBENCH_ROOT", str(repo)) + monkeypatch.setenv("MIRACLE_IFELSE_DIR", str(repo)) + # fresh import: drop any cached broken partial module so env stubs take effect + sys.modules.pop("miracle_matrix", None) + import importlib + return importlib.import_module("miracle_matrix") + + +def test_cli_run_resume_invokes_verify_and_proceeds(tmp_path, monkeypatch): + """verify 通过 -> resume() 被调用 + matrix.full.log 被写;execute() 不跑真实比赛。""" + mm = _import_cli_module(monkeypatch) + sid = "fake_sid_via_cli" + sd = tmp_path / sid + sd.mkdir(parents=True) + (sd / "manifest.json").write_text( + json.dumps({"run_id": "RID", "plan_count": 32, "plan": _PLAN32, + "timeout": 8.0, "wrapper_timeout_s": 180.0, + "session_id": sid}, ensure_ascii=False), encoding="utf-8") + (sd / "progress.json").write_text('{"attempts":{}}', encoding="utf-8") + monkeypatch.setattr(mm, "SESSION_ROOT", tmp_path) + + verified = [] + def fake_verify(sdp, **kw): + verified.append(str(sdp)) + return (True, []) + monkeypatch.setattr(mm, "verify_session_for_resume", fake_verify) + + resumed = {"count": 0} + executed = {"count": 0} + + class FakeRunner: + def __init__(self): + self.session_dir = sd + self.run_id = "RID" + self.session_id = sid + + def resume(self, s): + resumed["count"] += 1 + self.session_id = s + + def execute(self): + executed["count"] += 1 + return {"completed": True, "halted": False} + + def write_run_compatible_output(self): + return sd + + def aggregate_from_events(self): + return {"total_attempts": 0, "valid_games": 0, + "invalid_games": 0, "win_rate": None} + + roots = {name: tmp_path / name for name in ( + "judge_dir", "ifelse_dir", "extracted_root", "archives_root", + "precheck_root", "rank16_build_root", + )} + monkeypatch.setattr(mm, "verify_hashes", lambda *_a, **_k: []) + rc = mm._run_resume(FakeRunner(), sid, **roots) + assert rc == 0 + assert len(verified) == 1, "verify_session_for_resume must be called exactly once" + assert resumed["count"] == 1, "r.resume() must be called after verify OK" + assert executed["count"] == 1 + # matrix.full.log appended (resume path proceeds past verify) + log_path = sd / "matrix.full.log" + assert log_path.exists(), "matrix.full.log not written after resume" + + +def test_cli_run_resume_no_resume_when_verify_fails(tmp_path, monkeypatch): + """verify 失败 -> 不调用 r.resume()/execute()/write,返回 2。""" + mm = _import_cli_module(monkeypatch) + sid = "fake_sid_fail" + sd = tmp_path / sid + sd.mkdir(parents=True) + (sd / "manifest.json").write_text( + json.dumps({"run_id": "RID", "session_id": sid}, ensure_ascii=False), encoding="utf-8") + (sd / "progress.json").write_text('{"attempts":{}}', encoding="utf-8") + monkeypatch.setattr(mm, "SESSION_ROOT", tmp_path) + monkeypatch.setattr(mm, "verify_session_for_resume", + lambda *a, **k: (False, ["synthetic mismatch"])) + + resumed = {"count": 0} + executed = {"count": 0} + + class FakeRunner: + def __init__(self): + self.session_dir = sd + self.run_id = "RID" + self.session_id = sid + + def resume(self, s): + resumed["count"] += 1 + + def execute(self): + executed["count"] += 1 + return {"completed": True} + + def write_run_compatible_output(self): + return sd + + def aggregate_from_events(self): + return {} + + roots = {name: tmp_path / name for name in ( + "judge_dir", "ifelse_dir", "extracted_root", "archives_root", + "precheck_root", "rank16_build_root", + )} + rc = mm._run_resume(FakeRunner(), sid, **roots) + assert rc == 2, "verify-fail must return 2" + assert resumed["count"] == 0, "must NOT resume when verify fails" + assert executed["count"] == 0 + assert not (sd / "matrix.full.log").exists(), "log must NOT be opened when verify fails" + + +# =========================================================================== # +# 9. resume() 后不再走宽松 load_progress(同一文件严格解析一致) +# =========================================================================== # +def test_resume_strict_rejects_corrupt_progress(tmp_path): + """验证通过后,manifest 没变、progress 立即被人为破坏成 NOT JSON —— 此时 + resume() 必须严格解析并拒绝,而不是宽松 load_progress 把它当成空 progress + 默默通过。""" + from agentbench_frame.games.miracle.matrix_runner import MatrixRunner + + r = MatrixRunner( + session_root=tmp_path, judge_dir=tmp_path / "j", ifelse_dir=tmp_path / "i", + opponent_dir_of=lambda rk: tmp_path / f"o{rk}", vendor_script=tmp_path / "v.py", + framework_src=tmp_path / "src", attempt_fn=lambda **k: None, + run_id="RID", + ) + r.prepare_session() + sid = r.session_id + # write a valid manifest (so a fresh verify_session_for_resume passes by itself) + r.record_manifest(opponent_hashes={i: "h" for i in range(1, 17)}, + build_hashes={i: "b" for i in range(1, 17)}, + ifelse_sha="IF", judge_sha="JD", code_hashes={}) + # corrupt progress (NOT JSON) + (r.progress_path).write_text("NOT JSON {{{", encoding="utf-8") + # resume() MUST raise on the strict parse — never silently return empty + with pytest.raises((ValueError, json.JSONDecodeError, RuntimeError)): + r2 = MatrixRunner( + session_root=tmp_path, judge_dir=tmp_path / "j", ifelse_dir=tmp_path / "i", + opponent_dir_of=lambda rk: tmp_path / f"o{rk}", vendor_script=tmp_path / "v.py", + framework_src=tmp_path / "src", attempt_fn=lambda **k: None, + run_id="RID2", + ) + r2.resume(sid) + + +def test_resume_strict_rejects_non_dict_progress(tmp_path): + from agentbench_frame.games.miracle.matrix_runner import MatrixRunner + r = MatrixRunner( + session_root=tmp_path, judge_dir=tmp_path / "j", ifelse_dir=tmp_path / "i", + opponent_dir_of=lambda rk: tmp_path / f"o{rk}", vendor_script=tmp_path / "v.py", + framework_src=tmp_path / "src", attempt_fn=lambda **k: None, run_id="RID", + ) + r.prepare_session() + sid = r.session_id + r.record_manifest(opponent_hashes={i: "h" for i in range(1, 17)}, + build_hashes={i: "b" for i in range(1, 17)}, + ifelse_sha="IF", judge_sha="JD", code_hashes={}) + # progress is a JSON array, not an object + (r.progress_path).write_text("[1, 2, 3]", encoding="utf-8") + with pytest.raises((ValueError, TypeError, RuntimeError)): + r2 = MatrixRunner( + session_root=tmp_path, judge_dir=tmp_path / "j", ifelse_dir=tmp_path / "i", + opponent_dir_of=lambda rk: tmp_path / f"o{rk}", vendor_script=tmp_path / "v.py", + framework_src=tmp_path / "src", attempt_fn=lambda **k: None, run_id="RID2", + ) + r2.resume(sid) + + +# =========================================================================== # +# 10. events.jsonl 非 JSON 对象行(int / str / 数组顶层)必须保守拒绝 +# =========================================================================== # +def test_events_non_json_object_line_rejected(tmp_path): + """events.jsonl 含一行 `42`(合法 JSON 但顶层不是对象)—— 必须报错,不能崩。""" + proc = {"attempts": {}} + _wm(tmp_path, progress=proc) + (tmp_path / "events.jsonl").write_text("42\n", encoding="utf-8") + ok, errs = verify_session_for_resume(tmp_path) + assert not ok + assert any("events" in e.lower() and + ("corrupt" in e.lower() or "object" in e.lower() or "invalid" in e.lower()) + for e in errs), errs + + +def test_events_string_top_level_rejected(tmp_path): + proc = {"attempts": {}} + _wm(tmp_path, progress=proc) + (tmp_path / "events.jsonl").write_text('"hello"\n', encoding="utf-8") + ok, errs = verify_session_for_resume(tmp_path) + assert not ok + assert any("events" in e.lower() for e in errs), errs + + +def test_events_array_top_level_rejected(tmp_path): + proc = {"attempts": {}} + _wm(tmp_path, progress=proc) + (tmp_path / "events.jsonl").write_text('[1, 2, 3]\n', encoding="utf-8") + ok, errs = verify_session_for_resume(tmp_path) + assert not ok + assert any("events" in e.lower() for e in errs), errs + + +def test_events_no_keys_object_with_no_game_id_is_acceptable(tmp_path): + """一个合法 JSON 对象但没有 game_id —— 不是 corrupt,不应崩,但因为无 game_id + 所以不影响 done/not_started 比对。只是确保 isinstance(e, dict) 保护生效,不报 corrupt。""" + proc = {"attempts": {}} + _wm(tmp_path, progress=proc) + (tmp_path / "events.jsonl").write_text('{"event":"meta","note":"x"}\n', encoding="utf-8") + ok, errs = verify_session_for_resume(tmp_path) + # 不应出现 events corrupt 错误 + assert not any("events" in e.lower() and "corrupt" in e.lower() for e in errs), errs + + +# =========================================================================== # +# 额外: done 局 attempt_fn 调用次数严格为 0;不生成第二个 run 目录 +# =========================================================================== # +def test_resume_done_games_call_attempt_fn_zero_times(tmp_path): + """session 中 rank01 两 camp done;resume 后 execute() 必须不调用 attempt_fn。""" + from agentbench_frame.games.miracle.matrix_runner import MatrixRunner + from agentbench_frame.games.miracle import matrix as mx + + calls = [] + def fake_fn(**kw): + calls.append(kw["game_id"]) + return None # should not be reached + + r = MatrixRunner( + session_root=tmp_path, judge_dir=tmp_path / "j", ifelse_dir=tmp_path / "i", + opponent_dir_of=lambda rk: tmp_path / f"o{rk}", vendor_script=tmp_path / "v.py", + framework_src=tmp_path / "src", attempt_fn=fake_fn, run_id="ORIGINAL_RID", + ) + r.prepare_session() + r.record_manifest(opponent_hashes={i: "h" + str(i) for i in range(1, 17)}, + build_hashes={i: "b" + str(i) for i in range(1, 17)}, + ifelse_sha="IF", judge_sha="JD", code_hashes={}) + mx.mark_done(r.progress, "m_rank01_camp0", _done_entry(rank=1, camp=0)) + mx.mark_done(r.progress, "m_rank01_camp1", _done_entry(rank=1, camp=1, norm="loss")) + mx.write_progress_atomic(r.progress_path, r.progress) + # events for the done games (audit requires 1 event each? no - audit checks + # events equivalence; let's write minimal events) + mx.append_event_atomic(r.events_path, + {"event": "game", "game_id": "m_rank01_camp0", + "valid": True, "normalized_result": "win"}) + mx.append_event_atomic(r.events_path, + {"event": "game", "game_id": "m_rank01_camp1", + "valid": True, "normalized_result": "loss"}) + # rank audit should exist for the done pair (so we don't trigger + # "complete rank missing audit" — we write a VALID audit) + audit_dir = r.session_dir / "audit" + audit_dir.mkdir(parents=True, exist_ok=True) + (audit_dir / "rank01.json").write_text(json.dumps({ + "rank": 1, "ok": True, "reasons": [], + "games": [{"game_id": "m_rank01_camp0", "camp": 0}, + {"game_id": "m_rank01_camp1", "camp": 1}], + }, ensure_ascii=False), encoding="utf-8") + + # new runner with DIFFERENT run_id -- resume must keep the original run_id + r2 = MatrixRunner( + session_root=tmp_path, judge_dir=tmp_path / "j", ifelse_dir=tmp_path / "i", + opponent_dir_of=lambda rk: tmp_path / f"o{rk}", vendor_script=tmp_path / "v.py", + framework_src=tmp_path / "src", attempt_fn=fake_fn, run_id="WRONG", + ) + r2.resume(r.session_id) + assert r2.run_id == "ORIGINAL_RID" + # only execute rank 1 (both camps done) — should not call attempt_fn + res = r2.execute_rank(rank=1) + assert res.get("halted") is False + assert calls == [], f"attempt_fn called {len(calls)} times for done games: {calls}" + # run_dir baked from ORIGINAL_RID — no second run dir created + assert "WRONG" not in str(r2.run_dir) + # session inventory unchanged (only ONE session dir under tmp_path) + sessions = [p for p in tmp_path.iterdir() if p.is_dir()] + assert len(sessions) == 1, f"resume created extra session dirs: {sessions}" diff --git a/tests/miracle/test_section4_full_chain.py b/tests/miracle/test_section4_full_chain.py new file mode 100644 index 0000000..4a126bd --- /dev/null +++ b/tests/miracle/test_section4_full_chain.py @@ -0,0 +1,683 @@ +"""Section 4 端到端贯通测试(MatchAttempt → GameOutcome → to_event_record → JSON)。 + +约束(与事前提上的契约一致): + +1. 不只构造最终 GameOutcome,必须经过包含完整 fake result-json 的 + ``_build_attempt_from_files`` 入口逐值验证。 +2. 验证以下字段在「result-json→MatchAttempt→GameOutcome→to_event_record→JSON 落盘重读」 + 全链逐字一致: + * vendor exception 原文(``vendor_exception``,逐字 result-json 的 ``exception``) + * wrapper exception 原文(``wrapper_exception``,外层 wrapper Popen/包装异常) + * 有 reason 但无真实 exception 时 ``exception`` MUST 为 null(不得用 reason 填) + * ``timeout``(result-json 原始 ``timeout`` 形态)与 ``timeout_s``(实际传入的逐步超时) + * Judge/ai0/ai1 不同 final_returncode + * ``process_cleanup`` 包含四个 role,且来源明确 + (vendor = ProcessTreeManager 外层;judge/ai0/ai1 = result-json 内层) + * cleanup failure 走分类 ``cleanup_failure`` + * 内部 ``run_match_returncode`` 与外部 ``vendor_returncode`` 各自落事件且不混淆 + * Replay SHA-256 + * ``evidence_paths`` 全套保留 +""" +from __future__ import annotations + +import json +import os +import struct +import sys +from pathlib import Path + +import pytest + +REPO = Path(__file__).resolve().parents[2] +SRC = REPO / "src" + + +# --------------------------------------------------------------------------- # +# fake result-json shape (mirrors vendor/miracle_local/run_match.py output) +# --------------------------------------------------------------------------- # +def _inner_status(*, role, pid, final_returncode, **over): + """Per-role status block the vendor writes into result-json.""" + s = { + "role": role, "pid": pid, "started_at": 1000.0, + "natural_exit": False, "natural_returncode": None, + "termination_requested": False, "termination_reason": None, + "final_returncode": final_returncode, + "forced_kill": False, "cleanup_succeeded": True, "identity_confirmed": True, + } + s.update(over) + return s + + +def _fake_result_json(tmp_path, tag, *, vendor_exception=None, + cleanup_all_succeeded=True, run_match_returncode=0, + end_info_received=True, end_info=None, + raw_winner=0, timeout_flag=None, ai_error=None, + ai0_final_returncode=0, ai1_final_returncode=0, + judge_final_returncode=0, scores=None, + score_tie=False): + if timeout_flag is None: + timeout_flag = {"ai0": False, "ai1": False} + if ai_error is None: + ai_error = {"ai0": False, "ai1": False} + if end_info is None: + end_info = {"0": 5, "1": 2} + if scores is None: + scores = {"0": 5, "1": 2} + return { + "schema_version": 1, "tag": str(tag), + "started_at": 1000.0, "finished_at": 1005.0, "duration_s": 5.0, + "judge_dir_resolved": str(tmp_path / "fake_judge"), + "p0": {"name": "ifelse", "dir": str(tmp_path / "p0")}, + "p1": {"name": "rank01", "dir": str(tmp_path / "p1")}, + "judge": _inner_status(role="judge", pid=111, final_returncode=judge_final_returncode, + natural_exit=True, natural_returncode=judge_final_returncode), + "ai0": _inner_status(role="ai0", pid=112, final_returncode=ai0_final_returncode), + "ai1": _inner_status(role="ai1", pid=113, final_returncode=ai1_final_returncode), + "end_info_received": end_info_received, + "end_info": end_info, + "scores": scores, + "raw_winner": raw_winner, + "score_tie": score_tie, + "judge_tiebreak_applied": score_tie, + "timeout": timeout_flag, + "ai_error": ai_error, + "trace_path": str(tmp_path / "work" / f"{tag}.jsonl"), + "replay_path": str(tmp_path / "work" / f"{tag}.replay"), + "cleanup_all_succeeded": cleanup_all_succeeded, + "exception": vendor_exception, + "run_match_returncode": run_match_returncode, + } + + +def _vendor_manager_status(*, vendor_final_rc=-1, cleanup_succeeded=True, + forced_kill=False, termination_requested=True, + termination_reason="wrapper-timeout"): + return [{ + "role": "vendor", "pid": 999, "started_at": 1000.0, + "natural_exit": False, "natural_returncode": None, + "termination_requested": termination_requested, + "termination_reason": termination_reason, + "final_returncode": vendor_final_rc, + "forced_kill": forced_kill, + "cleanup_succeeded": cleanup_succeeded, + "identity_confirmed": True, + }] + + + + + +# =========================================================================== # +# 1. Main full-chain propagation test (vendor exception + wrapper exception + +# timeout/timeout_s + per-role exits + 4-role cleanup + internal/external rc +# + replay SHA + evidence_paths + reason without exception) +# =========================================================================== # +def test_full_chain_field_propagation_via_fake_result_json(tmp_path): + from agentbench_frame.games.miracle.match_runner import _build_attempt_from_files + from agentbench_frame.games.miracle.runner import attempt_to_outcome + from agentbench_frame.games.miracle.result import to_event_record + from agentbench_frame.games.miracle.atomicio import atomic_write_json + from agentbench_frame.games.miracle.result import sha256_file + + tag = "m_rank01_camp0" + work_dir = tmp_path / "work" + work_dir.mkdir(parents=True, exist_ok=True) + + vendor_exception_text = "RuntimeError('vendor AI read EOF before end')" + payload = _fake_result_json( + tmp_path, tag, + vendor_exception=vendor_exception_text, + cleanup_all_succeeded=False, + run_match_returncode=7, + ai0_final_returncode=1, ai1_final_returncode=0, judge_final_returncode=0, + timeout_flag={"ai0": False, "ai1": True}, + raw_winner=0, end_info={"0": 5, "1": 2}, scores={"0": 5, "1": 2}, + ) + pop = payload.pop # local alias + atomic_write_json(work_dir / f"{tag}.result.json", payload) + ei = json.dumps({"0": 5, "1": 2}) + (work_dir / f"{tag}.jsonl").write_text( + json.dumps({"kind": "ai_operation", "player": 0}) + "\n" + + json.dumps({"kind": "match_end", "end_info": ei}) + "\n", + encoding="utf-8", + ) + replay_bytes = struct.pack(">7i", 0, 0, 0, 1, 0, 0, 0) + b"\x00" * 16 + replay_path = work_dir / f"{tag}.replay" + replay_path.write_bytes(replay_bytes) + (work_dir / f"{tag}.stdout").write_text("out", encoding="utf-8") + (work_dir / f"{tag}.stderr").write_text("err", encoding="utf-8") + + wrapper_exception_text = "OSError('wrapper Popen failed: phantom')" + mgr_status = _vendor_manager_status(vendor_final_rc=-1, cleanup_succeeded=True, + forced_kill=True, + termination_reason="wrapper-timeout") + + att = _build_attempt_from_files( + game_id=tag, evaluated_agent="ifelse", opponent="rank01", + evaluated_agent_camp=0, + work_dir=work_dir, tag=tag, + vendor_returncode=-1, wrapper_timeout=True, + wrapper_exception=wrapper_exception_text, + vendor_manager_status=mgr_status, + timeout_s=8.0, started=999.0, finished=1006.0, + collision_detected=False, + ) + + # === MatchAttempt 级断言 === + assert att.vendor_exception == vendor_exception_text, "vendor_exception must be verbatim from result-json" + assert att.wrapper_exception == wrapper_exception_text, "wrapper_exception must be verbatim outer Popen exc" + assert att.timeout == {"ai0": False, "ai1": True}, "timeout must be the raw result-json timeout dict" + assert att.timeout_s == 8.0, "timeout_s must be the run_match_attempt timeout kwarg" + assert att.wrapper_timeout is True + assert att.judge_exit == 0 + assert att.ai0_exit == 1 + assert att.ai1_exit == 0 + assert att.run_match_returncode == 7 + assert att.vendor_returncode == -1 + assert att.replay_sha256 == sha256_file(replay_path) + assert att.reason and isinstance(att.reason, str) + # exception compat rule: wrapper_exception 拥有最高优先级,无 reason 填充 + assert att.exception == wrapper_exception_text, \ + "exception compat field must equal wrapper_exception (NOT reason)" + # evidence_paths 5 项保留 + assert set(att.evidence_paths.keys()) == {"stdout", "stderr", "trace", "replay", "result_json"} + + # process_cleanup 必须含四个 role + source 标记 + roles = sorted(r["role"] for r in att.process_cleanup) + assert roles == ["ai0", "ai1", "judge", "vendor"], \ + f"process_cleanup must contain 4 roles, got {roles}" + sources = {(r["role"], r["source"]) for r in att.process_cleanup} + assert ("vendor", "process_tree_manager") in sources, \ + "vendor must be tagged source=process_tree_manager" + assert ("judge", "result_json") in sources + assert ("ai0", "result_json") in sources + assert ("ai1", "result_json") in sources + + # === GameOutcome 级断言 === + o = attempt_to_outcome(att) + assert o.vendor_exception == vendor_exception_text + assert o.wrapper_exception == wrapper_exception_text + assert o.timeout == {"ai0": False, "ai1": True} + assert o.timeout_s == 8.0 + assert o.run_match_returncode == 7 + assert o.judge_exit == 0 and o.ai0_exit == 1 and o.ai1_exit == 0 + assert o.vendor_returncode == -1 + assert o.replay_sha256 == att.replay_sha256 + assert o.exception == wrapper_exception_text + assert o.reason and o.reason == att.reason + + # === to_event_record + JSON 落盘重读 === + rec = to_event_record(o) + rec_path = tmp_path / "sample_event.jsonl" + rec_path.write_text(json.dumps(rec, default=str, ensure_ascii=False) + "\n", + encoding="utf-8") + loaded = json.loads(rec_path.read_text(encoding="utf-8").strip()) + + assert loaded["timeout"] == {"ai0": False, "ai1": True} + assert loaded["timeout_s"] == 8.0 + assert loaded["wrapper_timeout"] is True + assert loaded["wrapper_exception"] == wrapper_exception_text + assert loaded["vendor_exception"] == vendor_exception_text + assert loaded["run_match_returncode"] == 7 + assert loaded["vendor_returncode"] == -1 + assert loaded["judge_exit"] == 0 + assert loaded["ai0_exit"] == 1 + assert loaded["ai1_exit"] == 0 + assert loaded["result_json_status"] == "ok" + assert loaded["reason"] == att.reason + # exception is null when no real exception — but here we have a real exception, + # so it must equal wrapper_exception (the highest priority real exception source) + assert loaded["exception"] == wrapper_exception_text + # internal rc NOT silently feeding reason / exception: + assert "RuntimeError" not in loaded.get("reason", "") # reason doesn't echo exception text + # replay sha same as the MatchAttempt's hash: + assert loaded["replay_sha256"] == att.replay_sha256 + # evidence_paths preserved on the event layer too: + assert loaded["evidence_paths"]["replay"].endswith(".replay") + assert loaded["evidence_paths"]["result_json"].endswith(".result.json") + + +# =========================================================================== # +# 2. reason populated but no real exception → exception MUST be null +# =========================================================================== # +def test_reason_without_real_exception_yields_null_exception(tmp_path): + """result-json MISSING case: classify produces a populated reason, but + there is no real exception (no wrapper exception, no vendor exception), + so the exception (compat field) MUST be null — never filled by reason.""" + from agentbench_frame.games.miracle.match_runner import _build_attempt_from_files + from agentbench_frame.games.miracle.runner import attempt_to_outcome + from agentbench_frame.games.miracle.result import to_event_record + + tag = "m_rank02_camp0" + work_dir = tmp_path / "work" + work_dir.mkdir(parents=True, exist_ok=True) + # do NOT write a result-json → load_result_json returns ("missing", None) + (work_dir / f"{tag}.jsonl").write_text("", encoding="utf-8") # empty trace + (work_dir / f"{tag}.stdout").write_text("out", encoding="utf-8") + (work_dir / f"{tag}.stderr").write_text("err", encoding="utf-8") + # no replay file + + att = _build_attempt_from_files( + game_id=tag, evaluated_agent="ifelse", opponent="rank02", + evaluated_agent_camp=0, + work_dir=work_dir, tag=tag, + vendor_returncode=0, wrapper_timeout=False, + wrapper_exception=None, # no wrapper exception + vendor_manager_status=_vendor_manager_status( + vendor_final_rc=0, cleanup_succeeded=True, + termination_requested=False, termination_reason=None, + ), + timeout_s=12.0, started=1000.0, finished=1001.0, collision_detected=False, + ) + + # classify flags result_json_missing with a populated reason + assert att.error_type == "result_json_missing" + assert att.reason, "classify must produce a reason for the missing case" + assert att.reason and att.reason is not None + # but there is NO real exception anywhere → compat exception MUST be null + assert att.exception is None, "no real exception → compat exception MUST be None" + assert att.wrapper_exception is None + assert att.vendor_exception is None + assert att.valid is False + + o = attempt_to_outcome(att) + assert o.exception is None + assert o.reason and o.reason == att.reason + + rec = to_event_record(o) + rec_path = tmp_path / "ev2.jsonl" + rec_path.write_text(json.dumps(rec, default=str, ensure_ascii=False), + encoding="utf-8") + loaded = json.loads(rec_path.read_text(encoding="utf-8")) + assert loaded["exception"] is None, "exception sentinel must be null in JSON" + # reason is still preserved verbatim — proves exception wasn't filled by reason + assert loaded["reason"] == att.reason + assert loaded["error_type"] == "result_json_missing" + + +# =========================================================================== # +# 3. vendor_exception only (no wrapper exception) — compat exception = vendor_exception +# =========================================================================== # +def test_vendor_exception_only_path(tmp_path): + from agentbench_frame.games.miracle.match_runner import _build_attempt_from_files + from agentbench_frame.games.miracle.runner import attempt_to_outcome + from agentbench_frame.games.miracle.result import to_event_record + + tag = "m_rank03_camp1" + work_dir = tmp_path / "work" + work_dir.mkdir(parents=True, exist_ok=True) + from agentbench_frame.games.miracle.atomicio import atomic_write_json + vendor_exc = "ValueError('vendor side bad JSON')" + payload = _fake_result_json( + tmp_path, tag, + vendor_exception=vendor_exc, + cleanup_all_succeeded=True, run_match_returncode=1, + ai0_final_returncode=0, ai1_final_returncode=0, judge_final_returncode=0, + raw_winner=0, end_info={"0": 5, "1": 2}, scores={"0": 5, "1": 2}, + ) + atomic_write_json(work_dir / f"{tag}.result.json", payload) + ei = json.dumps({"0": 5, "1": 2}) + (work_dir / f"{tag}.jsonl").write_text( + json.dumps({"kind": "ai_operation", "player": 0}) + "\n" + + json.dumps({"kind": "match_end", "end_info": ei}) + "\n", + encoding="utf-8", + ) + (work_dir / f"{tag}.replay").write_bytes( + struct.pack(">7i", 0, 0, 0, 1, 1, 0, 0) + b"\x00" * 16) + (work_dir / f"{tag}.stdout").write_text("out", encoding="utf-8") + (work_dir / f"{tag}.stderr").write_text("err", encoding="utf-8") + + att = _build_attempt_from_files( + game_id=tag, evaluated_agent="ifelse", opponent="rank03", + evaluated_agent_camp=1, + work_dir=work_dir, tag=tag, + vendor_returncode=0, wrapper_timeout=False, + wrapper_exception=None, + vendor_manager_status=_vendor_manager_status( + vendor_final_rc=0, cleanup_succeeded=True, + termination_requested=False, termination_reason=None, + ), + timeout_s=8.0, started=1000.0, finished=1002.0, collision_detected=False, + ) + assert att.vendor_exception == vendor_exc + assert att.wrapper_exception is None + # compat exception rule: wrapper_exception (None) OR vendor_exception → vendor_exception + assert att.exception == vendor_exc, "compat exception must promote vendor_exception when no wrapper exception" + o = attempt_to_outcome(att) + rec = to_event_record(o) + p = tmp_path / "ev3.jsonl" + p.write_text(json.dumps(rec, default=str, ensure_ascii=False), encoding="utf-8") + loaded = json.loads(p.read_text(encoding="utf-8")) + assert loaded["vendor_exception"] == vendor_exc + assert loaded["wrapper_exception"] is None + assert loaded["exception"] == vendor_exc + + +# =========================================================================== # +# 4. cleanup failure classification path (no wrapper_timeout, both rcs 0, no exc) +# =========================================================================== # +def test_cleanup_failure_classification_path(tmp_path): + from agentbench_frame.games.miracle.match_runner import _build_attempt_from_files + from agentbench_frame.games.miracle.runner import attempt_to_outcome + from agentbench_frame.games.miracle.result import to_event_record + + tag = "m_rank04_camp0" + work_dir = tmp_path / "work" + work_dir.mkdir(parents=True, exist_ok=True) + from agentbench_frame.games.miracle.atomicio import atomic_write_json + payload = _fake_result_json( + tmp_path, tag, + vendor_exception=None, + cleanup_all_succeeded=False, + run_match_returncode=0, + ai0_final_returncode=0, ai1_final_returncode=0, judge_final_returncode=0, + raw_winner=0, end_info={"0": 5, "1": 2}, scores={"0": 5, "1": 2}, + ) + atomic_write_json(work_dir / f"{tag}.result.json", payload) + ei = json.dumps({"0": 5, "1": 2}) + (work_dir / f"{tag}.jsonl").write_text( + json.dumps({"kind": "ai_operation", "player": 0}) + "\n" + + json.dumps({"kind": "match_end", "end_info": ei}) + "\n", + encoding="utf-8", + ) + (work_dir / f"{tag}.replay").write_bytes( + struct.pack(">7i", 0, 0, 0, 0, 0, 0, 0) + b"\x00" * 16) + (work_dir / f"{tag}.stdout").write_text("out", encoding="utf-8") + (work_dir / f"{tag}.stderr").write_text("err", encoding="utf-8") + + att = _build_attempt_from_files( + game_id=tag, evaluated_agent="ifelse", opponent="rank04", + evaluated_agent_camp=0, + work_dir=work_dir, tag=tag, + vendor_returncode=0, wrapper_timeout=False, + wrapper_exception=None, + vendor_manager_status=_vendor_manager_status( + vendor_final_rc=0, cleanup_succeeded=False, + termination_requested=True, termination_reason="post_end_info", + ), + timeout_s=8.0, started=1000.0, finished=1005.0, collision_detected=False, + ) + assert att.error_type == "cleanup_failure", f"cleanup_failure not classified, got {att.error_type}: {att.reason}" + assert att.valid is False + o = attempt_to_outcome(att) + rec = to_event_record(o) + p = tmp_path / "ev4.jsonl" + p.write_text(json.dumps(rec, default=str, ensure_ascii=False), encoding="utf-8") + loaded = json.loads(p.read_text(encoding="utf-8")) + assert loaded["error_type"] == "cleanup_failure" + assert loaded["valid"] is False + # cleanup failed role vendor still in process_cleanup, sourced from mgr + assert any(r.get("role") == "vendor" and r.get("cleanup_succeeded") is False + for r in loaded["process_cleanup"]) + + +# =========================================================================== # +# 5. internal run_match_returncode lands separately from external vendor_returncode +# =========================================================================== # +def test_internal_run_match_returncode_lands_on_event_separately(tmp_path): + from agentbench_frame.games.miracle.match_runner import _build_attempt_from_files + from agentbench_frame.games.miracle.runner import attempt_to_outcome + from agentbench_frame.games.miracle.result import to_event_record + from agentbench_frame.games.miracle.atomicio import atomic_write_json + + tag = "m_rank05_camp0" + work_dir = tmp_path / "work" + work_dir.mkdir(parents=True, exist_ok=True) + payload = _fake_result_json( + tmp_path, tag, + vendor_exception=None, + cleanup_all_succeeded=True, run_match_returncode=3, + ai0_final_returncode=0, ai1_final_returncode=0, judge_final_returncode=0, + raw_winner=0, end_info={"0": 5, "1": 2}, scores={"0": 5, "1": 2}, + ) + atomic_write_json(work_dir / f"{tag}.result.json", payload) + ei = json.dumps({"0": 5, "1": 2}) + (work_dir / f"{tag}.jsonl").write_text( + json.dumps({"kind": "ai_operation", "player": 0}) + "\n" + + json.dumps({"kind": "match_end", "end_info": ei}) + "\n", + encoding="utf-8", + ) + (work_dir / f"{tag}.replay").write_bytes( + struct.pack(">7i", 0, 0, 0, 1, 1, 0, 0) + b"\x00" * 16) + (work_dir / f"{tag}.stdout").write_text("out", encoding="utf-8") + (work_dir / f"{tag}.stderr").write_text("err", encoding="utf-8") + + att = _build_attempt_from_files( + game_id=tag, evaluated_agent="ifelse", opponent="rank05", + evaluated_agent_camp=0, + work_dir=work_dir, tag=tag, + vendor_returncode=0, # outer wrapper returned 0 cleanly + wrapper_timeout=False, wrapper_exception=None, + vendor_manager_status=_vendor_manager_status( + vendor_final_rc=0, cleanup_succeeded=True, + termination_requested=False, termination_reason=None, + ), + timeout_s=8.0, started=1000.0, finished=1005.0, collision_detected=False, + ) + # classify must flag vendor_exception (internal rc != 0) and stop the matrix + assert att.error_type == "vendor_exception", att.reason + # the external rc stays 0, internal stays 3 (must not be conflated) + o = attempt_to_outcome(att) + rec = to_event_record(o) + p = tmp_path / "ev5.jsonl" + p.write_text(json.dumps(rec, default=str, ensure_ascii=False), encoding="utf-8") + loaded = json.loads(p.read_text(encoding="utf-8")) + assert loaded["run_match_returncode"] == 3 + assert loaded["vendor_returncode"] == 0 # both keys present and distinct: + assert "run_match_returncode" in loaded + assert "vendor_returncode" in loaded + + +# =========================================================================== # +# 6. evidence_paths 落事件且 result_json 路径以 .result.json 结尾 +# =========================================================================== # +def test_evidence_paths_lands_on_event(tmp_path): + from agentbench_frame.games.miracle.match_runner import _build_attempt_from_files + from agentbench_frame.games.miracle.runner import attempt_to_outcome + from agentbench_frame.games.miracle.result import to_event_record + from agentbench_frame.games.miracle.atomicio import atomic_write_json + + tag = "m_rank06_camp1" + work_dir = tmp_path / "work" + work_dir.mkdir(parents=True, exist_ok=True) + payload = _fake_result_json( + tmp_path, tag, + vendor_exception=None, cleanup_all_succeeded=True, run_match_returncode=0, + ai0_final_returncode=0, ai1_final_returncode=0, judge_final_returncode=0, + raw_winner=1, end_info={"0": 2, "1": 5}, scores={"0": 2, "1": 5}, + ) + atomic_write_json(work_dir / f"{tag}.result.json", payload) + ei = json.dumps({"0": 2, "1": 5}) + (work_dir / f"{tag}.jsonl").write_text( + json.dumps({"kind": "ai_operation", "player": 0}) + "\n" + + json.dumps({"kind": "match_end", "end_info": ei}) + "\n", + encoding="utf-8", + ) + (work_dir / f"{tag}.replay").write_bytes( + struct.pack(">7i", 0, 0, 0, 1, 0, 0, 0) + b"\x00" * 16) + (work_dir / f"{tag}.stdout").write_text("out", encoding="utf-8") + (work_dir / f"{tag}.stderr").write_text("err", encoding="utf-8") + + att = _build_attempt_from_files( + game_id=tag, evaluated_agent="ifelse", opponent="rank06", + evaluated_agent_camp=1, + work_dir=work_dir, tag=tag, + vendor_returncode=0, wrapper_timeout=False, wrapper_exception=None, + vendor_manager_status=_vendor_manager_status( + vendor_final_rc=0, cleanup_succeeded=True, + termination_requested=False, termination_reason=None, + ), + timeout_s=8.0, started=1000.0, finished=1001.0, collision_detected=False, + ) + o = attempt_to_outcome(att) + rec = to_event_record(o) + assert all(k in rec["evidence_paths"] for k in + ("stdout", "stderr", "trace", "replay", "result_json")) + assert rec["evidence_paths"]["result_json"].endswith(f"{tag}.result.json") + assert rec["evidence_paths"]["replay"].endswith(f"{tag}.replay") + + +# =========================================================================== # +# 7. timeout field shape — even when both ai flags are False, raw result-json +# timeout dict must be preserved verbatim on the event +# =========================================================================== # +def test_timeout_field_raw_dict_preserved(tmp_path): + from agentbench_frame.games.miracle.match_runner import _build_attempt_from_files + from agentbench_frame.games.miracle.runner import attempt_to_outcome + from agentbench_frame.games.miracle.result import to_event_record + from agentbench_frame.games.miracle.atomicio import atomic_write_json + + tag = "m_rank07_camp0" + work_dir = tmp_path / "work" + work_dir.mkdir(parents=True, exist_ok=True) + custom_timeout = {"ai0": True, "ai1": False} + payload = _fake_result_json( + tmp_path, tag, + vendor_exception=None, cleanup_all_succeeded=True, run_match_returncode=0, + ai0_final_returncode=2, ai1_final_returncode=0, judge_final_returncode=0, + timeout_flag=custom_timeout, + raw_winner=0, end_info={"0": 5, "1": 2}, scores={"0": 5, "1": 2}, + ) + atomic_write_json(work_dir / f"{tag}.result.json", payload) + ei = json.dumps({"0": 5, "1": 2}) + (work_dir / f"{tag}.jsonl").write_text( + json.dumps({"kind": "ai_operation", "player": 0}) + "\n" + + json.dumps({"kind": "ai_timeout", "player": 0}) + "\n" + + json.dumps({"kind": "match_end", "end_info": ei}) + "\n", + encoding="utf-8", + ) + (work_dir / f"{tag}.replay").write_bytes( + struct.pack(">7i", 0, 0, 0, 0, 1, 0, 0) + b"\x00" * 16) + (work_dir / f"{tag}.stdout").write_text("out", encoding="utf-8") + (work_dir / f"{tag}.stderr").write_text("err", encoding="utf-8") + + att = _build_attempt_from_files( + game_id=tag, evaluated_agent="ifelse", opponent="rank07", + evaluated_agent_camp=0, + work_dir=work_dir, tag=tag, + vendor_returncode=0, wrapper_timeout=False, wrapper_exception=None, + vendor_manager_status=_vendor_manager_status( + vendor_final_rc=0, cleanup_succeeded=True, + termination_requested=False, termination_reason=None, + ), + timeout_s=11.0, started=1000.0, finished=1002.0, collision_detected=False, + ) + assert att.timeout == custom_timeout # raw verbatim from result-json + assert att.timeout_s == 11.0 + o = attempt_to_outcome(att) + rec = to_event_record(o) + p = tmp_path / "ev7.jsonl" + p.write_text(json.dumps(rec, default=str, ensure_ascii=False), encoding="utf-8") + loaded = json.loads(p.read_text(encoding="utf-8")) + assert loaded["timeout"] == custom_timeout + assert loaded["timeout_s"] == 11.0 + + +# =========================================================================== # +# 8. boundary: _build_attempt_from_files must work even with vendor_manager_status=[] +# (no vendor process; only inner judge/ai0/ai1 from result-json land in process_cleanup) +# =========================================================================== # +def test_process_cleanup_handles_empty_vendor_manager_status(tmp_path): + from agentbench_frame.games.miracle.match_runner import _build_attempt_from_files + from agentbench_frame.games.miracle.atomicio import atomic_write_json + + tag = "m_rank08_camp0" + work_dir = tmp_path / "work" + work_dir.mkdir(parents=True, exist_ok=True) + payload = _fake_result_json( + tmp_path, tag, + vendor_exception=None, cleanup_all_succeeded=True, run_match_returncode=0, + ai0_final_returncode=0, ai1_final_returncode=0, judge_final_returncode=0, + raw_winner=0, end_info={"0": 5, "1": 2}, scores={"0": 5, "1": 2}, + ) + atomic_write_json(work_dir / f"{tag}.result.json", payload) + ei = json.dumps({"0": 5, "1": 2}) + (work_dir / f"{tag}.jsonl").write_text( + json.dumps({"kind": "ai_operation", "player": 0}) + "\n" + + json.dumps({"kind": "match_end", "end_info": ei}) + "\n", + encoding="utf-8", + ) + (work_dir / f"{tag}.replay").write_bytes( + struct.pack(">7i", 0, 0, 0, 0, 0, 0, 0) + b"\x00" * 16) + (work_dir / f"{tag}.stdout").write_text("out", encoding="utf-8") + (work_dir / f"{tag}.stderr").write_text("err", encoding="utf-8") + + att = _build_attempt_from_files( + game_id=tag, evaluated_agent="ifelse", opponent="rank08", + evaluated_agent_camp=0, + work_dir=work_dir, tag=tag, + vendor_returncode=0, wrapper_timeout=False, wrapper_exception=None, + vendor_manager_status=[], # empty — only result-json inner procs come in + timeout_s=8.0, started=1000.0, finished=1001.0, collision_detected=False, + ) + roles = sorted(r["role"] for r in att.process_cleanup) + assert roles == ["ai0", "ai1", "judge"] # no vendor, only inner 3 + # each comes from result-json + assert all(r["source"] == "result_json" for r in att.process_cleanup) + + +# =========================================================================== # +# 9. compat exception rule is deterministic & test-covered: +# wrapper_exception + vendor_exception both present → wrapper wins (no overwrite) +# wrapper=None, vendor=None → None +# wrapper set, vendor None → wrapper_exception +# wrapper None, vendor set → vendor_exception +# =========================================================================== # +def test_compat_exception_rule_deterministic(tmp_path): + from agentbench_frame.games.miracle.match_runner import _build_attempt_from_files + from agentbench_frame.games.miracle.atomicio import atomic_write_json + + def _build_for(*, wrapper_exception, vendor_exception): + tag = f"test_{abs(hash((wrapper_exception, vendor_exception))) % 100000}" + # use unique tag per call within the same tmp_path dir + work_dir = tmp_path / f"wd_{tag}" + work_dir.mkdir(parents=True, exist_ok=True) + payload = _fake_result_json( + tmp_path, tag, + vendor_exception=vendor_exception, + cleanup_all_succeeded=(vendor_exception is None), + run_match_returncode=0 if vendor_exception is None else 1, + ai0_final_returncode=0, ai1_final_returncode=0, judge_final_returncode=0, + timeout_flag={"ai0": False, "ai1": False}, + raw_winner=0, end_info={"0": 5, "1": 2}, scores={"0": 5, "1": 2}, + ) + atomic_write_json(work_dir / f"{tag}.result.json", payload) + ei = json.dumps({"0": 5, "1": 2}) + (work_dir / f"{tag}.jsonl").write_text( + json.dumps({"kind": "ai_operation", "player": 0}) + "\n" + + json.dumps({"kind": "match_end", "end_info": ei}) + "\n", + encoding="utf-8", + ) + (work_dir / f"{tag}.replay").write_bytes( + struct.pack(">7i", 0, 0, 0, 0, 0, 0, 0) + b"\x00" * 16) + (work_dir / f"{tag}.stdout").write_text("o", encoding="utf-8") + (work_dir / f"{tag}.stderr").write_text("e", encoding="utf-8") + att = _build_attempt_from_files( + game_id=tag, evaluated_agent="ifelse", opponent="opp", + evaluated_agent_camp=0, + work_dir=work_dir, tag=tag, + vendor_returncode=(0 if wrapper_exception is None else -1), + wrapper_timeout=(wrapper_exception is not None), + wrapper_exception=wrapper_exception, + vendor_manager_status=_vendor_manager_status( + vendor_final_rc=0, cleanup_succeeded=True, + termination_requested=(wrapper_exception is not None), + termination_reason="wrapper-timeout" if wrapper_exception else None, + ), + timeout_s=8.0, started=1000.0, finished=1002.0, collision_detected=False, + ) + return att + + # case 1: wrapper + vendor → wrapper wins + att = _build_for(wrapper_exception="WrapperErr('w')", vendor_exception="VendorErr('v')") + assert att.wrapper_exception == "WrapperErr('w')" and att.vendor_exception == "VendorErr('v')" + assert att.exception == "WrapperErr('w')" + # case 2: both None → None + att = _build_for(wrapper_exception=None, vendor_exception=None) + assert att.exception is None + # case 3: wrapper alone → wrapper exception + att = _build_for(wrapper_exception="WrapperErr('w2')", vendor_exception=None) + assert att.exception == "WrapperErr('w2')" + # case 4: vendor alone → vendor_exception + att = _build_for(wrapper_exception=None, vendor_exception="VendorErr('v2')") + assert att.exception == "VendorErr('v2')" \ No newline at end of file diff --git a/tools/miracle_matrix.py b/tools/miracle_matrix.py new file mode 100644 index 0000000..b8576a3 --- /dev/null +++ b/tools/miracle_matrix.py @@ -0,0 +1,614 @@ +#!/usr/bin/env python3 +"""24_miracle A-plan 32-game formal matrix runner (single process). + +Binds match_runner + matrix.py + matrix_runner + Framework Run/events/summary. +Run under ONE Python 3.11.15 ``uv run`` process (constant env). Modes: + --dry-run : create session + manifest + verify hashes + print plan; NO game/subprocess. + (default) : verify, then execute() the 32 attempts (per-rank audit, infra-stop), + then write Run-compatible output + aggregate. + +Opponent runnable dirs (frozen builds, NOT recompiled): + rank04/05/07 (Python): protected extracted dirs + rank01/02/03/06/08-15 (C++): .smoke/precheck/20260721-184652_da899d/strategies/rankNN + rank16 (C++): .smoke/rank16build/20260721-195738_baff71/rank16_copy +""" +from __future__ import annotations + +import hashlib +import argparse +import json +import platform +import sys +import time +from dataclasses import dataclass +from pathlib import Path +from typing import Sequence + +REPO = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(REPO / "src")) + +from agentbench_frame.games.miracle.matrix_runner import ( # noqa: E402 + MatrixRunner, + verify_session_for_resume, +) +from agentbench_frame.games.miracle.matrix import make_attempt_plan # noqa: E402 + +from agentbench_frame.games.miracle.paths import judge_dir, ifelse_dir, extracted_dir, archives_dir + + +def _safe_default(factory, fallback: Path) -> Path: + try: + return factory() + except RuntimeError: + return fallback + + +JUDGE = _safe_default(judge_dir, REPO / ".external" / "judge") +IFELSE = _safe_default(ifelse_dir, REPO / ".external" / "ifelse") +EXTRACTED = _safe_default(extracted_dir, REPO / ".external" / "extracted") +ARCHIVES = _safe_default(archives_dir, REPO / ".external" / "archives") +VENDOR = REPO / "vendor" / "miracle_local" / "run_match.py" +FW_SRC = REPO / "src" +PROTOCOL = REPO / "docs" / "games" / "24_miracle_evaluation_protocol.v0.3.json" +ROSTER = REPO / "docs" / "games" / "24_miracle_roster_manifest.json" +PRECHECK_9A = REPO / ".smoke" / "precheck" / "20260721-184652_da899d" +RANK16_BUILD = REPO / ".smoke" / "rank16build" / "20260721-195738_baff71" +SESSION_ROOT = REPO / ".smoke" / "matrix" +PROTOCOL_SHA = "f64b948c3dfef1e59d0d2db9ea747ed1d742d054bafd9abe6e6b824d65723b99" +AUTH_TEXT = "用户授权 A 方案 32 局正式矩阵(v0.3);逐对手分批审计;成功局不重跑;infra 即停;完成后聚合+本地 Results 验收;不 push/不上传。" + +PYTHON_RANKS = (4, 5, 7) +CPP_RANKS = (1, 2, 3, 6, 8, 9, 10, 11, 12, 13, 14, 15, 16) +LOG = None + + +class PreflightError(RuntimeError): + """A stable, user-facing input validation failure.""" + + +@dataclass(frozen=True) +class ControlInputs: + protocol: dict + roster: dict + hashes: dict[str, str] + + +def parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: + parser = argparse.ArgumentParser(description="Run the 24_miracle evaluation matrix") + parser.add_argument("--dry-run", action="store_true", help="plan only; do not start a game") + parser.add_argument("--resume", metavar="SESSION_ID", help="resume an existing session") + parser.add_argument("--protocol", type=Path, default=PROTOCOL) + parser.add_argument("--roster", type=Path, default=ROSTER) + parser.add_argument("--protocol-sha", default=None) + parser.add_argument("--session-root", type=Path, default=SESSION_ROOT) + parser.add_argument("--judge-dir", type=Path, default=JUDGE) + parser.add_argument("--ifelse-dir", type=Path, default=IFELSE) + parser.add_argument("--extracted-root", type=Path, default=EXTRACTED) + parser.add_argument("--archives-root", type=Path, default=ARCHIVES) + parser.add_argument("--precheck-root", type=Path, default=PRECHECK_9A) + parser.add_argument("--rank16-build-root", type=Path, default=RANK16_BUILD) + return parser.parse_args(argv) + + +def log(msg=""): + print(msg, flush=True) + if LOG: + LOG.write(str(msg) + "\n"); LOG.flush() + + +def sha(p: Path) -> str: + h = hashlib.sha256() + with open(p, "rb") as f: + for b in iter(lambda: f.read(1 << 20), b""): + h.update(b) + return h.hexdigest() + + +def control_text_sha(path: Path) -> str: + """Hash UTF-8 control text after normalizing physical newlines to LF.""" + text = Path(path).read_text(encoding="utf-8") + normalized = text.replace("\r\n", "\n").replace("\r", "\n") + return hashlib.sha256(normalized.encode("utf-8")).hexdigest() + + +def _load_json_object(path: Path, label: str) -> dict: + if not path.exists(): + raise PreflightError(f"{label} file missing: {path}") + try: + value = json.loads(path.read_text(encoding="utf-8")) + except Exception as exc: + raise PreflightError(f"{label} JSON invalid: {path}: {exc}") from exc + if not isinstance(value, dict): + raise PreflightError(f"{label} JSON must be an object: {path}") + return value + + +def _require_control_mapping(value, label: str) -> dict: + if not isinstance(value, dict): + raise PreflightError(f"control schema: {label} missing or not an object") + return value + + +def _require_control_sha(value, label: str) -> str: + if ( + not isinstance(value, str) + or len(value) != 64 + or any(char not in "0123456789abcdef" for char in value) + ): + raise PreflightError(f"control schema: {label} missing or not a 64-character lowercase hexadecimal SHA256") + return value + + +def _validate_control_schema(protocol: dict, strategies: list[dict]) -> None: + """Reject incomplete control inputs before a runner or session exists.""" + identities = _require_control_mapping(protocol.get("frozen_identities"), "frozen_identities") + evaluated_agent = _require_control_mapping( + identities.get("evaluated_agent"), "frozen_identities.evaluated_agent" + ) + _require_control_sha( + evaluated_agent.get("sha256"), "frozen_identities.evaluated_agent.sha256" + ) + judge = _require_control_mapping(identities.get("judge"), "frozen_identities.judge") + _require_control_sha(judge.get("main_py_sha256"), "frozen_identities.judge.main_py_sha256") + builds = _require_control_mapping( + identities.get("build_artifacts_win64_mingw"), + "frozen_identities.build_artifacts_win64_mingw", + ) + for rank in CPP_RANKS: + _require_control_sha( + builds.get(f"rank{rank:02d}"), + f"frozen_identities.build_artifacts_win64_mingw.rank{rank:02d}", + ) + for strategy in strategies: + rank = strategy["rank"] + label = f"roster.strategies.rank{rank:02d}" + _require_control_sha(strategy.get("archive_sha256"), f"{label}.archive_sha256") + if rank in PYTHON_RANKS: + entry = strategy.get("entry") + if not isinstance(entry, str) or not entry: + raise PreflightError(f"control schema: {label}.entry missing or not a string") + _require_control_sha(strategy.get("runnable_sha256"), f"{label}.runnable_sha256") + + +def load_control_inputs( + protocol_path: Path, + roster_path: Path, + expected_protocol_sha: str | None = None, +) -> ControlInputs: + protocol_path = Path(protocol_path) + roster_path = Path(roster_path) + protocol = _load_json_object(protocol_path, "protocol") + roster = _load_json_object(roster_path, "roster") + strategies = roster.get("strategies") + if not isinstance(strategies, list): + raise PreflightError("roster strategies must be a list") + if len(strategies) != 16 or any(not isinstance(item, dict) for item in strategies): + raise PreflightError("roster ranks must be exactly 1..16: invalid strategy entry") + ranks = [item.get("rank") for item in strategies if isinstance(item, dict)] + if ranks != list(range(1, 17)): + raise PreflightError(f"roster ranks must be exactly 1..16: {ranks}") + _validate_control_schema(protocol, strategies) + protocol_hash = control_text_sha(protocol_path) + if expected_protocol_sha and protocol_hash != expected_protocol_sha: + raise PreflightError( + f"protocol sha mismatch: got {protocol_hash}, expected {expected_protocol_sha}" + ) + return ControlInputs( + protocol=protocol, + roster=roster, + hashes={"protocol": protocol_hash, "roster": control_text_sha(roster_path)}, + ) + + +def resolve_unique_dir(root: Path, pattern: str) -> Path: + root = Path(root) + matches = sorted(path for path in root.glob(pattern) if path.is_dir()) + if not matches: + raise FileNotFoundError(f"no opponent directory matching {pattern}: {root}") + if len(matches) > 1: + names = ", ".join(str(path) for path in matches) + raise RuntimeError(f"multiple opponent directories matching {pattern}: {names}") + return matches[0] + + +def resolve_unique_file(root: Path, pattern: str) -> Path: + root = Path(root) + matches = sorted(path for path in root.glob(pattern) if path.is_file()) + if not matches: + raise FileNotFoundError(f"no archive file matching {pattern}: {root}") + if len(matches) > 1: + names = ", ".join(str(path) for path in matches) + raise RuntimeError(f"multiple archive files matching {pattern}: {names}") + return matches[0] + + +def resolve_opponent_dir( + rank: int, + roster: dict | None = None, + *, + extracted_root: Path = EXTRACTED, + precheck_root: Path = PRECHECK_9A, + rank16_build_root: Path = RANK16_BUILD, +) -> Path: + if rank in PYTHON_RANKS: + return resolve_unique_dir(extracted_root, f"rank{rank:02d}__*") + if rank == 16: + return Path(rank16_build_root) / "rank16_copy" + return Path(precheck_root) / "strategies" / f"rank{rank:02d}" + + +def opponent_dir_of(rank: int) -> Path: + return resolve_opponent_dir(rank) + + +def _strategy_label(strategy: dict) -> str: + rank = strategy.get("rank", "unknown") + return f"rank{int(rank):02d}" if isinstance(rank, int) else f"rank{rank}" + + +def verify_archive_hash(strategy: dict, archives_root: Path) -> list[str]: + """Verify the unique archive for one roster entry using raw-byte hashing.""" + rank = strategy.get("rank") + label = _strategy_label(strategy) + if not isinstance(rank, int): + return [f"{label}: invalid roster rank"] + try: + archive = resolve_unique_file(Path(archives_root), f"rank{rank:02d}__*.zip") + except (FileNotFoundError, RuntimeError) as exc: + return [f"{label}: {exc}"] + expected_archive_sha = strategy.get("archive_sha256") + if not expected_archive_sha: + return [f"{label}: archive sha missing from roster"] + try: + actual_archive_sha = sha(archive) + except OSError as exc: + return [f"{label}: archive sha unreadable: {exc}"] + if actual_archive_sha != expected_archive_sha: + return [f"{label}: archive sha mismatch"] + return [] + + +def verify_python_strategy_hashes(strategy: dict, extracted_root: Path) -> list[str]: + """Verify a Python opponent's extracted runnable entry only. + + Archives are verified once for every roster entry by ``verify_archive_hash``. + """ + rank = strategy.get("rank", "unknown") + label = _strategy_label(strategy) + errors: list[str] = [] + try: + directory = resolve_unique_dir(extracted_root, f"rank{int(rank):02d}__*") + except (FileNotFoundError, RuntimeError) as exc: + return [f"{label}: {exc}"] + entry = strategy.get("entry") + if not isinstance(entry, str) or not entry or Path(entry).is_absolute() or ".." in Path(entry).parts: + return [f"{label}: invalid runnable entry"] + entry_path = directory / entry + if not entry_path.is_file(): + errors.append(f"{label}: runnable entry missing: {entry_path}") + else: + expected_entry_sha = strategy.get("runnable_sha256") + if not expected_entry_sha: + errors.append(f"{label}: runnable sha missing from roster") + else: + try: + actual_entry_sha = sha(entry_path) + except OSError as exc: + errors.append(f"{label}: runnable sha unreadable: {exc}") + else: + if actual_entry_sha != expected_entry_sha: + errors.append(f"{label}: runnable sha mismatch") + return errors + + +def verify_hashes( + v3: dict, + roster: dict, + *, + extracted_root: Path = EXTRACTED, + archives_root: Path = ARCHIVES, + precheck_root: Path = PRECHECK_9A, + rank16_build_root: Path = RANK16_BUILD, + ifelse_root: Path = IFELSE, + judge_root: Path = JUDGE, + asset_digests: dict[str, str] | None = None, +): + """Verify every opponent's runnable identity matches the frozen hashes.""" + mismatches = [] + ba = v3["frozen_identities"]["build_artifacts_win64_mingw"] + strategies = {item["rank"]: item for item in roster.get("strategies", []) if isinstance(item, dict)} + for rank in range(1, 17): + strategy = strategies.get(rank) + if strategy is None: + mismatches.append(f"rank{rank:02d}: roster strategy missing") + else: + mismatches.extend(verify_archive_hash(strategy, archives_root)) + try: + d = resolve_opponent_dir( + rank, + roster, + extracted_root=extracted_root, + precheck_root=precheck_root, + rank16_build_root=rank16_build_root, + ) + except (FileNotFoundError, RuntimeError) as exc: + mismatches.append(f"rank{rank:02d}: {exc}") + continue + if not d.exists(): + mismatches.append(f"rank{rank:02d}: opponent dir missing {d}") + continue + if rank in PYTHON_RANKS: + if strategy is None: + continue + else: + mismatches.extend(verify_python_strategy_hashes(strategy, extracted_root)) + else: + me = d / "main.exe" + if not me.is_file(): + mismatches.append(f"rank{rank:02d}: main.exe missing") + else: + want = ba[f"rank{rank:02d}"] + try: + got = sha(me) + except OSError as exc: + mismatches.append(f"rank{rank:02d}: main.exe sha unreadable: {exc}") + else: + if got != want: + mismatches.append(f"rank{rank:02d}: main.exe sha {got} != frozen {want}") + ifelse_main = Path(ifelse_root) / "main.py" + if not ifelse_main.is_file(): + mismatches.append(f"ifelse main.py missing: {ifelse_main}") + else: + try: + ifelse_sha = sha(ifelse_main) + except OSError as exc: + mismatches.append(f"ifelse sha unreadable: {exc}") + else: + if ifelse_sha != v3["frozen_identities"]["evaluated_agent"]["sha256"]: + mismatches.append("ifelse sha mismatch") + elif asset_digests is not None: + asset_digests["ifelse"] = ifelse_sha + judge_main = Path(judge_root) / "main.py" + if not judge_main.is_file(): + mismatches.append(f"judge main.py missing: {judge_main}") + else: + try: + judge_sha = sha(judge_main) + except OSError as exc: + mismatches.append(f"judge sha unreadable: {exc}") + else: + if judge_sha != v3["frozen_identities"]["judge"]["main_py_sha256"]: + mismatches.append("judge sha mismatch") + elif asset_digests is not None: + asset_digests["judge"] = judge_sha + return mismatches + + +def _run_resume( + r, + resume_sid, + *, + control_inputs: ControlInputs | None = None, + session_root: Path | None = None, + judge_dir: Path | None = None, + ifelse_dir: Path | None = None, + extracted_root: Path | None = None, + archives_root: Path | None = None, + precheck_root: Path | None = None, + rank16_build_root: Path | None = None, +): + """Resume an existing session. VERIFY FIRST (read-only); only if ALL checks + pass, call resume() + write. Verification failure leaves session untouched.""" + global LOG + session_root = Path(session_root or SESSION_ROOT) + sd = session_root / resume_sid + if not sd.exists(): + print(f"FATAL: session not found: {sd}", file=sys.stderr) + return 2 + # READ-ONLY verification — load protocol + roster for full identity + if control_inputs is None: + control_inputs = load_control_inputs(PROTOCOL, ROSTER, PROTOCOL_SHA) + v3 = control_inputs.protocol + roster = control_inputs.roster + ba = v3["frozen_identities"]["build_artifacts_win64_mingw"] + ok, errs = verify_session_for_resume( + sd, + protocol_sha=control_inputs.hashes["protocol"], + code_files={ + "matrix": str(REPO / "src/agentbench_frame/games/miracle/matrix.py"), + "matrix_runner": str(REPO / "src/agentbench_frame/games/miracle/matrix_runner.py"), + "match_runner": str(REPO / "src/agentbench_frame/games/miracle/match_runner.py"), + "vendor_run_match": str(VENDOR), + }, + expected_plan=make_attempt_plan(), + expected_ifelse_sha=v3["frozen_identities"]["evaluated_agent"]["sha256"], + expected_judge_sha=v3["frozen_identities"]["judge"]["main_py_sha256"], + expected_opponent_shas={s["rank"]: s["archive_sha256"] for s in roster["strategies"]}, + expected_build_shas={int(k.replace("rank", "")): v for k, v in ba.items() + if k.startswith("rank") and isinstance(v, str) and "_" not in k}, + expected_platform=platform.platform(), + expected_control_inputs=control_inputs.hashes, + ) + if not ok: + print("FATAL: resume verification failed:", file=sys.stderr) + for e in errs: + print(f" - {e}", file=sys.stderr) + return 2 + runtime_roots = ( + ("judge_dir", judge_dir), ("ifelse_dir", ifelse_dir), + ("extracted_root", extracted_root), ("archives_root", archives_root), + ("precheck_root", precheck_root), ("rank16_build_root", rank16_build_root), + ) + missing_roots = [name for name, value in runtime_roots if value is None] + if missing_roots: + print(f"FATAL: resume runtime root missing: {', '.join(missing_roots)}", file=sys.stderr) + return 2 + mismatches = verify_hashes( + v3, roster, + extracted_root=Path(extracted_root), archives_root=Path(archives_root), + precheck_root=Path(precheck_root), rank16_build_root=Path(rank16_build_root), + ifelse_root=Path(ifelse_dir), judge_root=Path(judge_dir), + ) + if mismatches: + print("FATAL: resume current asset verification failed:", file=sys.stderr) + for mismatch in mismatches: + print(f" - {mismatch}", file=sys.stderr) + return 2 + # only now: open session + write + r.resume(resume_sid) + LOG = open(r.session_dir / "matrix.full.log", "a", encoding="utf-8") + log(f"RESUME session: {resume_sid} (verified)") + result = r.execute() + log(f"resume result: {json.dumps({k: v for k, v in result.items() if k != 'record'}, ensure_ascii=False)}") + r.write_run_compatible_output() + agg = r.aggregate_from_events() + log(f"aggregate: total={agg['total_attempts']} valid={agg['valid_games']} invalid={agg['invalid_games']} win_rate={agg['win_rate']}") + return 0 if result.get("completed") else 1 + + +def validate_runtime_paths(*, judge_root: Path, ifelse_root: Path, extracted_root: Path, + archives_root: Path, precheck_root: Path, + rank16_build_root: Path) -> None: + for label, path in ( + ("judge directory", judge_root), + ("ifelse directory", ifelse_root), + ("extracted opponent directory", extracted_root), + ("opponent archives directory", archives_root), + ("precheck directory", precheck_root), + ("rank16 build directory", rank16_build_root), + ): + if not Path(path).is_dir(): + raise PreflightError(f"{label} missing: {path}") + for label, path in (("judge main.py", Path(judge_root) / "main.py"), + ("ifelse main.py", Path(ifelse_root) / "main.py")): + if not path.is_file(): + raise PreflightError(f"{label} missing: {path}") + + +def main(argv: Sequence[str] | None = None) -> int: + global LOG + args = parse_args(argv) + if args.dry_run and args.resume: + print("FATAL: --resume and --dry-run are mutually exclusive", file=sys.stderr) + return 2 + expected_protocol_sha = args.protocol_sha + if expected_protocol_sha is None and args.protocol == PROTOCOL: + expected_protocol_sha = PROTOCOL_SHA + try: + control_inputs = load_control_inputs(args.protocol, args.roster, expected_protocol_sha) + validate_runtime_paths( + judge_root=args.judge_dir, + ifelse_root=args.ifelse_dir, + extracted_root=args.extracted_root, + archives_root=args.archives_root, + precheck_root=args.precheck_root, + rank16_build_root=args.rank16_build_root, + ) + except PreflightError as exc: + print(f"FATAL: {exc}", file=sys.stderr) + return 2 + v3 = control_inputs.protocol + roster = control_inputs.roster + + def opponent_resolver(rank: int) -> Path: + return resolve_opponent_dir( + rank, + roster, + extracted_root=args.extracted_root, + precheck_root=args.precheck_root, + rank16_build_root=args.rank16_build_root, + ) + + r = MatrixRunner( + session_root=args.session_root, judge_dir=args.judge_dir, ifelse_dir=args.ifelse_dir, + opponent_dir_of=opponent_resolver, vendor_script=VENDOR, framework_src=FW_SRC, + timeout=8.0, wrapper_timeout_s=180.0, + protocol_sha=control_inputs.hashes["protocol"], + python=sys.executable, evaluated_agent="miracle_ifelse", auth_text=AUTH_TEXT, + ) + if args.resume: + return _run_resume( + r, + args.resume, + control_inputs=control_inputs, + session_root=args.session_root, + judge_dir=args.judge_dir, + ifelse_dir=args.ifelse_dir, + extracted_root=args.extracted_root, + archives_root=args.archives_root, + precheck_root=args.precheck_root, + rank16_build_root=args.rank16_build_root, + ) + r.prepare_session() + LOG = open(r.session_dir / "matrix.full.log", "w", encoding="utf-8") + log(f"session_id: {r.session_id}") + log(f"run_id: {r.run_id}") + log(f"python: {sys.executable} ({platform.python_version()})") + log(f"protocol v0.3 sha: {control_inputs.hashes['protocol']} (verified)") + log(f"mode: {'DRY_RUN' if args.dry_run else 'EXECUTE'}") + + asset_digests: dict[str, str] = {} + mismatches = verify_hashes( + v3, + roster, + extracted_root=args.extracted_root, + archives_root=args.archives_root, + precheck_root=args.precheck_root, + rank16_build_root=args.rank16_build_root, + ifelse_root=args.ifelse_dir, + judge_root=args.judge_dir, + asset_digests=asset_digests, + ) + if mismatches: + log("FATAL: hash mismatches -> STOP before any game:") + for m in mismatches: + log(" - " + m) + return 2 + ifelse_sha = asset_digests.get("ifelse") + judge_sha = asset_digests.get("judge") + if ifelse_sha is None or judge_sha is None: + log("FATAL: verified ifelse/judge digest missing -> STOP before any game") + return 2 + opp_arch = {s["rank"]: s["archive_sha256"] for s in roster["strategies"]} + cpp_build = {int(k.replace("rank", "")): v for k, v in v3["frozen_identities"]["build_artifacts_win64_mingw"].items() + if k.startswith("rank") and isinstance(v, str) and "_" not in k} + r.record_manifest( + opponent_hashes=opp_arch, build_hashes=cpp_build, + ifelse_sha=ifelse_sha, + judge_sha=judge_sha, + code_hashes={"matrix": sha(REPO / "src/agentbench_frame/games/miracle/matrix.py"), + "matrix_runner": sha(REPO / "src/agentbench_frame/games/miracle/matrix_runner.py"), + "match_runner": sha(REPO / "src/agentbench_frame/games/miracle/match_runner.py"), + "vendor_run_match": sha(VENDOR)}, + control_inputs={ + "protocol": {"path": str(args.protocol), "sha256": control_inputs.hashes["protocol"]}, + "roster": {"path": str(args.roster), "sha256": control_inputs.hashes["roster"]}, + }, + ) + log(f"manifest: {r.session_dir / 'manifest.json'}") + log("hash verification: ALL_MATCH") + + if args.dry_run: + out = r.dry_run() + log(f"plan: {out['plan_count']} attempts; first={out['plan'][0]['game_id']} last={out['plan'][-1]['game_id']}") + log(f"timeout={r.timeout} wrapper_timeout_s={r.wrapper_timeout_s}") + log("DRY_RUN complete — no game/subprocess started.") + return 0 + + log("\n===== EXECUTE 32 attempts (single process) =====") + t0 = time.time() + result = r.execute() + log(f"execute result: {json.dumps({k: v for k, v in result.items() if k != 'record'}, ensure_ascii=False)}") + log(f"elapsed: {round(time.time() - t0, 1)}s") + run_dir = r.write_run_compatible_output() + log(f"Run-compatible output: {run_dir}") + agg = r.aggregate_from_events() + log(f"aggregate: total={agg['total_attempts']} valid={agg['valid_games']} invalid={agg['invalid_games']} " + f"wins={agg['wins']} losses={agg['losses']} win_rate={agg['win_rate']}") + (r.session_dir / "aggregate.json").write_text(json.dumps(agg, ensure_ascii=False, indent=2), encoding="utf-8") + log(f"aggregate.json: {r.session_dir / 'aggregate.json'}") + return 0 if result.get("completed") else 1 + + +if __name__ == "__main__": + raise SystemExit(main())