{ "bench_version": 1, "track": "dynamics", "model": "rnn_small", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": { "lr": 0.006, "degree": 2 }, "sweep": [ { "cfg": { "lr": 0.001, "degree": 2 }, "mean": 0.003609913313994184 }, { "cfg": { "lr": 0.003, "degree": 2 }, "mean": 0.004211013438180089 }, { "cfg": { "lr": 0.006, "degree": 2 }, "mean": 0.0017769129190128297 }, { "cfg": { "lr": 0.001, "degree": 3 }, "mean": 0.02138834842480719 }, { "cfg": { "lr": 0.003, "degree": 3 }, "mean": 0.005644599266815931 }, { "cfg": { "lr": 0.006, "degree": 3 }, "mean": 0.0025558681518305093 } ], "full": { "mean": 0.0015814606886124238, "std": 0.0004392499281827401, "per_seed": [ 0.0010742004960775375, 0.0021969457156956196, 0.0015373978530988097, 0.002299107611179352, 0.001275557209737599, 0.0012933816760778427, 0.0011743184877559543, 0.0018007764592766762 ], "n": 8 } }, "idea": { "mean": 0.0021942039020359516, "std": 0.0007557853669569652, "per_seed": [ 0.0022852420806884766, 0.002649907488375902, 0.001380756264552474, 0.003865908132866025, 0.001558732008561492, 0.0023790807463228703, 0.0016722833970561624, 0.0017617210978642106 ], "n": 8, "sweep": [ { "cfg": { "lr": 0.001, "threshold": 0.15 }, "mean": 0.02138834842480719 }, { "cfg": { "lr": 0.001, "threshold": 0.25 }, "mean": 0.021395483752712607 }, { "cfg": { "lr": 0.001, "threshold": 0.4 }, "mean": 0.02141962433233857 }, { "cfg": { "lr": 0.003, "threshold": 0.15 }, "mean": 0.005644599266815931 }, { "cfg": { "lr": 0.003, "threshold": 0.25 }, "mean": 0.005646273668389767 }, { "cfg": { "lr": 0.003, "threshold": 0.4 }, "mean": 0.005661312199663371 }, { "cfg": { "lr": 0.006, "threshold": 0.15 }, "mean": 0.0025558681518305093 }, { "cfg": { "lr": 0.006, "threshold": 0.25 }, "mean": 0.0025564142561051995 }, { "cfg": { "lr": 0.006, "threshold": 0.4 }, "mean": 0.0025454534916207194 } ] }, "comparison": { "delta_mean": 0.0006127432134235278, "idea_wins": 2, "n_pairs": 8, "per_seed_diffs": [ 0.001211041584610939, 0.0004529617726802826, -0.0001566415885463357, 0.0015668005216866732, 0.00028317479882389307, 0.0010856990702450275, 0.0004979649093002081, -3.905536141246557e-05 ], "p_value": 0.0324, "mde": 0.0005150508833037192, "mde_rel_pct": 32.5680484511838, "verdict": "idea worse (significant)", "system_worked": false }, "mechanism_signature": { "confirmed": true, "prediction": "cubic residual proxy scales as amplitude^3", "observed_mean_proxy_ratio_at_1p5": 3.375, "mean_rho": 0.8967330753803253, "mean_active_fraction": 0.9953124970197678, "track_match": "dynamics/control pendulum; trained GRU systems evaluated on standard test MSE" }, "protocol_notes": "8 paired seeds; baseline and idea share LiftedGRU; 12 epochs, batch 128; all tried learning rates are in both grids." }