{ "bench_version": 1, "track": "tabular", "model": "mlp_tiny", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": { "lr": 0.006, "weight_decay": 0.0 }, "sweep": [ { "config": { "lr": 0.001, "weight_decay": 0.0 }, "results": [ 30.65192222595215, 26.219579696655273, 21.692813873291016, 23.84225082397461 ], "mean": 25.60164165496826 }, { "config": { "lr": 0.003, "weight_decay": 0.0 }, "results": [ 13.522106170654297, 14.461689949035645, 14.807339668273926, 12.651887893676758 ], "mean": 13.860755920410156 }, { "config": { "lr": 0.006, "weight_decay": 0.0 }, "results": [ 10.067484855651855, 12.066851615905762, 9.622394561767578, 10.014313697814941 ], "mean": 10.442761182785034 }, { "config": { "lr": 0.003, "weight_decay": 0.0001 }, "results": [ 13.528329849243164, 14.431048393249512, 14.758162498474121, 12.640395164489746 ], "mean": 13.839483976364136 } ], "full": { "mean": 10.930935978889465, "std": 1.3983305692789296, "per_seed": [ 10.067484855651855, 12.066851615905762, 9.622394561767578, 10.014313697814941, 8.787652015686035, 11.810267448425293, 12.032318115234375, 13.046205520629883 ], "n": 8 } }, "idea": { "mean": 10.996043920516968, "std": 1.169447967118747, "per_seed": [ 9.497220993041992, 12.14220142364502, 11.333624839782715, 9.796889305114746, 9.323980331420898, 11.6688232421875, 11.8473482131958, 12.35826301574707 ], "n": 8 }, "comparison": { "delta_mean": 0.06510794162750244, "idea_wins": 5, "n_pairs": 8, "per_seed_diffs": [ -0.5702638626098633, 0.07534980773925781, 1.7112302780151367, -0.2174243927001953, 0.5363283157348633, -0.14144420623779297, -0.18496990203857422, -0.6879425048828125 ], "p_value": 0.8863, "mde": 0.6389629457421157, "mde_rel_pct": 5.845455018455167, "verdict": "no measurable effect", "system_worked": false }, "mechanism_signature": { "quantity": "mean squared action error to quadratic proximal teacher on held-out states", "predicted": "proximal objective should reduce teacher-action mismatch", "observed_note": "computed from trained models on benchmark-derived perturbations; full values recorded below", "baseline_best_cfg": { "lr": 0.006, "weight_decay": 0.0 }, "idea_best_cfg": { "lr": 0.006, "weight_decay": 0.0 }, "confirmed": false }, "idea_sweep": [ { "config": { "lr": 0.001, "weight_decay": 0.0 }, "full": { "mean": 25.094110250473022, "std": 4.455265918443395, "per_seed": [ 27.867267608642578, 23.728221893310547, 22.022008895874023, 23.82248878479004, 17.760637283325195, 31.66156578063965, 22.79692268371582, 31.093769073486328 ], "n": 8 }, "mean": 25.094110250473022 }, { "config": { "lr": 0.003, "weight_decay": 0.0 }, "full": { "mean": 14.121166110038757, "std": 1.3397338959449183, "per_seed": [ 13.135297775268555, 14.45909309387207, 14.704912185668945, 12.331259727478027, 12.489986419677734, 13.940347671508789, 15.557445526123047, 16.35098648071289 ], "n": 8 }, "mean": 14.121166110038757 }, { "config": { "lr": 0.006, "weight_decay": 0.0 }, "full": { "mean": 10.996043920516968, "std": 1.169447967118747, "per_seed": [ 9.497220993041992, 12.14220142364502, 11.333624839782715, 9.796889305114746, 9.323980331420898, 11.6688232421875, 11.8473482131958, 12.35826301574707 ], "n": 8 }, "mean": 10.996043920516968 } ], "protocol": { "paired_seeds": [ 0, 1, 2, 3, 4, 5, 6, 7 ], "epochs": 18, "batch": 128, "grid_union": [ { "lr": 0.001, "weight_decay": 0.0 }, { "lr": 0.003, "weight_decay": 0.0 }, { "lr": 0.006, "weight_decay": 0.0 }, { "lr": 0.003, "weight_decay": 0.0001 } ], "structural_match": "tabular: loss/regularization intervention" } }