Subcritical Percolation Jordan Readout / report_bench_2026-08-31T160824.md
Mechanism confirmed, baseline not beaten
Стенд-проверка (stage-2) · промт оператора:
(универсальный)
Ответ агента:
{ "worked": false, "confidence": 10, "verdict": "The official benchmark was run on the registered custom:csbm_radius_graph track using the canonical bench training and reporting functions. JordanNet improved mean error from 0.1184375 to 0.1093750, but the paired permutation p-value was 0.2165, so the required significant-win verdict was not reached. The mechanism signature was confirmed, but the system did not establish a reliable benchmark win.", "metrics": { "baseline": "Best baseline lr=0.003, gamma=0.0; full 8-seed mean error 0.11843749694526196; sweep over 9 configurations.", "idea": "Best JordanNet lr=0.003, gamma=0.5; 8-seed mean error 0.10937499720603228; paired delta=-0.009062499739229679; 4/8 paired wins; p=0.2165." }, "bench_report": { "bench_version": 1, "track": "custom:csbm_radius_graph", "model": "GraphNet", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": { "lr": 0.003, "gamma": 0.0 }, "sweep": [ { "cfg": { "lr": 0.001, "gamma": 0.0 }, "mean": 0.11812499724328518 }, { "cfg": { "lr": 0.001, "gamma": 0.5 }, "mean": 0.11812499724328518 }, { "cfg": { "lr": 0.001, "gamma": 0.8 }, "mean": 0.11812499724328518 }, { "cfg": { "lr": 0.003, "gamma": 0.0 }, "mean": 0.11499999649822712 }, { "cfg": { "lr": 0.003, "gamma": 0.5 }, "mean": 0.11499999649822712 }, { "cfg": { "lr": 0.003, "gamma": 0.8 }, "mean": 0.11499999649822712 }, { "cfg": { "lr": 0.01, "gamma": 0.0 }, "mean": 0.11562499590218067 }, { "cfg": { "lr": 0.01, "gamma": 0.5 }, "mean": 0.11562499590218067 }, { "cfg": { "lr": 0.01, "gamma": 0.8 }, "mean": 0.11562499590218067 } ], "full": { "mean": 0.11843749694526196, "std": 0.01745250183842921, "per_seed": [ 0.09999999403953552, 0.0949999988079071, 0.14249999821186066, 0.1224999949336052, 0.11249999701976776, 0.14749999344348907, 0.11749999970197678, 0.10999999940395355 ], "n": 8 } }, "idea": { "mean": 0.10937499720603228, "std": 0.01184205993180543, "per_seed": [ 0.10499999672174454, 0.11249999701976776, 0.11999999731779099, 0.10499999672174454, 0.08249999582767487, 0.11749999970197678, 0.1224999949336052, 0.10999999940395355 ], "n": 8 }, "comparison": { "delta_mean": -0.009062499739229679, "idea_wins": 4, "n_pairs": 8, "per_seed_diffs": [ 0.005000002682209015, 0.017499998211860657, -0.02250000089406967, -0.017499998211860657, -0.030000001192092896, -0.0299999937415123, 0.004999995231628418, 0.0 ], "p_value": 0.2165, "mde": 0.01519513610205344, "mde_rel_pct": 12.829666696752422, "verdict": "no significant win", "system_worked": false }, "mechanism_signature": { "predicted": "Jordan candidates are selected consistently across independent percolated views", "observed_candidate_count": 2.0, "observed_model_system": "trained JordanNet evaluated on the registered root-classification task", "confirmed": true }, "idea_grid": [ { "cfg": { "lr": 0.003, "gamma": 0.5 }, "mean": 0.10937499720603228, "std": 0.01184205993180543, "per_seed": [ 0.10499999672174454, 0.11249999701976776, 0.11999999731779099, 0.10499999672174454, 0.08249999582767487, 0.11749999970197678, 0.1224999949336052, 0.10999999940395355 ] }, { "cfg": { "lr": 0.001, "gamma": 0.5 }, "mean": 0.10937499720603228, "std": 0.011907849530802815, "per_seed": [ 0.10249999910593033, 0.11999999731779099, 0.11749999970197678, 0.10999999940395355, 0.08249999582767487, 0.10749999433755875, 0.1224999949336052, 0.11249999701976776 ] }, { "cfg": { "lr": 0.01, "gamma": 0.5 }, "mean": 0.1109374975785613, "std": 0.014786899093388256, "per_seed": [ 0.09999999403953552, 0.11999999731779099, 0.125, 0.09749999642372131, 0.08249999582767487, 0.11749999970197678, 0.1274999976158142, 0.11749999970197678 ] } ], "sanity": { "path_center": 3, "expected": 3, "passed": true } }, "how_to_run": "/home/maxwelhelp/main/bin/python3 stage2_bench.py", "files": [ "stage2_bench.py", "bench_report.json" ], "limitations": "Only the registered csbm_radius_graph distribution was tested, with fixed p=0.25, R=8, K=1, L=8 and 24 epochs. Larger graphs, alternate p/K/L sweeps, clean-versus-shortcut robustness, runtime scaling, and consistency-loss training were not tested.", "system_verdict": "partial", "practical_verdict": "inconclusive", "mechanism_ok": 1, "system_judged": true }