{ "bench_version": 1, "track": "tabular", "model": "mlp_tiny", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": { "lr": 0.003, "momentum": 0.9 }, "sweep": [ { "cfg": { "lr": 0.003, "momentum": 0.0 }, "mean": 11.063072919845581 }, { "cfg": { "lr": 0.003, "momentum": 0.9 }, "mean": 8.369025349617004 }, { "cfg": { "lr": 0.01, "momentum": 0.0 }, "mean": 27.549113154411316 }, { "cfg": { "lr": 0.01, "momentum": 0.9 }, "mean": 21.363237619400024 }, { "cfg": { "lr": 0.03, "momentum": 0.0 }, "mean": 24.645026683807373 }, { "cfg": { "lr": 0.03, "momentum": 0.9 }, "mean": 24.67656421661377 } ], "full": { "mean": 8.462080538272858, "std": 0.763596092727506, "per_seed": [ 9.01702880859375, 8.495253562927246, 8.408374786376953, 7.555444240570068, 7.235288619995117, 8.890181541442871, 9.808090209960938, 8.286982536315918 ], "n": 8 } }, "idea": { "mean": 11.567220091819763, "std": 1.9770478862735161, "per_seed": [ 11.564115524291992, 11.624547958374023, 11.304121017456055, 9.648195266723633, 8.412535667419434, 11.701919555664062, 12.67589282989502, 15.606432914733887 ], "n": 8 }, "comparison": { "delta_mean": 3.1051395535469055, "idea_wins": 0, "n_pairs": 8, "per_seed_diffs": [ 2.547086715698242, 3.1292943954467773, 2.8957462310791016, 2.0927510261535645, 1.1772470474243164, 2.8117380142211914, 2.867802619934082, 7.319450378417969 ], "p_value": 0.0081, "mde": 1.515669973753461, "mde_rel_pct": 17.91131586255045, "verdict": "idea worse (significant)", "system_worked": false }, "protocol_note": "baseline sweep uses all lr values tried by idea and sweeps momentum; idea uses three lr settings.", "mechanism_signature": { "prediction": "on easy well-conditioned tabular training adaptive order remains mostly p=0 while residual contracts", "observed_activation_fraction": 1.0, "per_seed": [ { "seed": 0, "activated": true, "first_activation": 5, "grad_ratio": 0.42842475212727243 }, { "seed": 1, "activated": true, "first_activation": 5, "grad_ratio": 0.3283609454108231 }, { "seed": 2, "activated": true, "first_activation": 5, "grad_ratio": 0.3528653471373228 }, { "seed": 3, "activated": true, "first_activation": 5, "grad_ratio": 0.2514142861201013 }, { "seed": 4, "activated": true, "first_activation": 5, "grad_ratio": 0.27445735401524274 }, { "seed": 5, "activated": true, "first_activation": 5, "grad_ratio": 0.2794134775570858 }, { "seed": 6, "activated": true, "first_activation": 5, "grad_ratio": 0.36328053123347775 }, { "seed": 7, "activated": true, "first_activation": 5, "grad_ratio": 0.3765249020584892 } ], "confirmed": false }, "idea_candidates": [ { "cfg": { "lr": 0.003, "momentum": 0.9 }, "result": { "mean": 11.567220091819763, "std": 1.9770478862735161, "per_seed": [ 11.564115524291992, 11.624547958374023, 11.304121017456055, 9.648195266723633, 8.412535667419434, 11.701919555664062, 12.67589282989502, 15.606432914733887 ], "n": 8 } }, { "cfg": { "lr": 0.01, "momentum": 0.9 }, "result": { "mean": 18.361396193504333, "std": 10.464071364766456, "per_seed": [ 41.432411193847656, 12.722975730895996, 24.072275161743164, 7.684035301208496, 11.690810203552246, 15.437067985534668, 24.278358459472656, 9.573235511779785 ], "n": 8 } }, { "cfg": { "lr": 0.03, "momentum": 0.9 }, "result": { "mean": 28.241891384124756, "std": 10.055774240316543, "per_seed": [ 52.05145263671875, 23.101425170898438, 16.589689254760742, 23.7571964263916, 29.247011184692383, 33.055755615234375, 23.479705810546875, 24.652894973754883 ], "n": 8 } } ] }