{ "bench_version": 1, "track": "tabular", "model": "variable_mlp", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": { "depth": 4, "lr": 0.006 }, "sweep": [ { "cfg": { "depth": 1, "lr": 0.0015 }, "mean": 179.61644744873047 }, { "cfg": { "depth": 1, "lr": 0.003 }, "mean": 61.134450912475586 }, { "cfg": { "depth": 1, "lr": 0.006 }, "mean": 19.637452602386475 }, { "cfg": { "depth": 2, "lr": 0.0015 }, "mean": 48.220858573913574 }, { "cfg": { "depth": 2, "lr": 0.003 }, "mean": 18.314094066619873 }, { "cfg": { "depth": 2, "lr": 0.006 }, "mean": 15.357292652130127 }, { "cfg": { "depth": 3, "lr": 0.0015 }, "mean": 24.715059757232666 }, { "cfg": { "depth": 3, "lr": 0.003 }, "mean": 16.38268733024597 }, { "cfg": { "depth": 3, "lr": 0.006 }, "mean": 12.54684829711914 }, { "cfg": { "depth": 4, "lr": 0.0015 }, "mean": 22.451377391815186 }, { "cfg": { "depth": 4, "lr": 0.003 }, "mean": 14.139441013336182 }, { "cfg": { "depth": 4, "lr": 0.006 }, "mean": 11.652069807052612 } ], "full": { "per_seed": [ 10.326997756958008, 11.77970027923584, 10.673856735229492, 11.692980766296387, 9.679404258728027, 11.413150787353516, 11.966154098510742, 13.212504386901855 ], "mean": 11.343093633651733, "std": 1.098638538871677 } }, "idea": { "per_seed": [ 10.326997756958008, 11.77970027923584, 10.673856735229492, 11.692980766296387, 9.679404258728027, 11.413150787353516, 11.966154098510742, 13.212504386901855 ], "mean": 11.343093633651733, "std": 1.098638538871677 }, "comparison": { "delta_mean": 0.0, "idea_wins": 0, "n_pairs": 8, "per_seed_diffs": [ 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0 ], "p_value": 1.0, "mde": 0.0, "mde_rel_pct": 0.0, "verdict": "no measurable effect", "system_worked": false }, "mechanism_signature": { "predicted_radius_depth_exponent": -1.0, "observed_test_error_depth_slope": -0.38086177111636127, "predicted_vs_observed_tolerance": 0.35, "confirmed": false, "observed_depth_errors": [ [ 1, 18.856087684631348 ], [ 2, 13.66954231262207 ], [ 3, 12.318396091461182 ], [ 4, 11.053349018096924 ] ], "note": "Measured on independently trained benchmark models." }, "bench_report": { "track": "tabular", "model": "variable_mlp", "calibration": { "floor_L": -19.786727674992903, "floor_U": 5.600807448724559, "C_syn": 70.65927382249097, "rows": [ [ 1, 66.96246337890625 ], [ 2, 17.991180419921875 ], [ 3, 16.63127326965332 ], [ 4, 17.250062942504883 ] ], "residual_std": 6.346883780929366 }, "chosen_depth": 4 } }