{ "bench_version": 1, "track": "dynamics", "model": "rnn_small", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "sweep": { "best_cfg": { "lr": 0.003, "rounds": 4 }, "sweep": [ { "cfg": { "lr": 0.001, "rounds": 4 }, "mean": 0.062015012837946415 }, { "cfg": { "lr": 0.003, "rounds": 4 }, "mean": 0.02783486293628812 }, { "cfg": { "lr": 0.01, "rounds": 4 }, "mean": 0.031387015245854855 } ], "full": { "mean": 0.03436466213315725, "std": 0.012433180922138218, "per_seed": [ 0.028132088482379913, 0.009857414290308952, 0.046546123921871185, 0.02681129425764084, 0.028106732293963432, 0.046415776014328, 0.045219533145427704, 0.043828334659338 ], "n": 8 } }, "best_cfg": { "lr": 0.003, "rounds": 4 }, "full": { "mean": 0.03436466213315725, "std": 0.012433180922138218, "per_seed": [ 0.028132088482379913, 0.009857414290308952, 0.046546123921871185, 0.02681129425764084, 0.028106732293963432, 0.046415776014328, 0.045219533145427704, 0.043828334659338 ], "n": 8 } }, "idea": { "mean": 0.0322596225887537, "std": 0.01605094247073741, "per_seed": [ 0.008163033053278923, 0.04895344749093056, 0.02211311273276806, 0.046254031360149384, 0.026844650506973267, 0.018853165209293365, 0.028555940836668015, 0.05833959951996803 ], "n": 8 }, "comparison": { "delta_mean": -0.002105039544403553, "idea_wins": 5, "n_pairs": 8, "per_seed_diffs": [ -0.01996905542910099, 0.039096033200621605, -0.024433011189103127, 0.019442737102508545, -0.0012620817869901657, -0.027562610805034637, -0.01666359230875969, 0.014511264860630035 ], "p_value": 0.80015, "mde": 0.020276508456807906, "mde_rel_pct": 59.003951146791046, "verdict": "no significant win", "system_worked": false }, "protocol_note": "Matched dynamics task and freshly trained identical rnn_small architecture; only the inference stopping rule differs.", "mechanism_signature": { "quantity": "NN observed disagreement contraction", "predicted_mean_ratio": 0.29535447060123415, "observed_mean_ratio": 1.0081205979778152, "absolute_error": 0.7127661273765811, "n_trajectories": 8, "confirmed": false }, "idea_sweep": [ { "cfg": { "lr": 0.001, "rounds": 4 }, "mean": 0.06841936893761158 }, { "cfg": { "lr": 0.003, "rounds": 4 }, "mean": 0.034365283441729844 }, { "cfg": { "lr": 0.01, "rounds": 4 }, "mean": 0.0322564683156088 } ], "idea_best_cfg": { "lr": 0.01, "rounds": 4 }, "idea_auxiliary_rounds": [ { "rounds": 4, "changes": [ 0.026034874841570854, 0.025428546592593193, 0.02514001540839672 ] }, { "rounds": 4, "changes": [ 0.0409514456987381, 0.042881350964307785, 0.04315655678510666 ] }, { "rounds": 4, "changes": [ 0.03505569323897362, 0.03519539535045624, 0.03512731194496155 ] }, { "rounds": 4, "changes": [ 0.042443498969078064, 0.043852899223566055, 0.04409702122211456 ] }, { "rounds": 4, "changes": [ 0.03753160312771797, 0.038068048655986786, 0.03855843096971512 ] }, { "rounds": 4, "changes": [ 0.03447379544377327, 0.034757938235998154, 0.034541524946689606 ] }, { "rounds": 4, "changes": [ 0.03614768758416176, 0.03720034658908844, 0.03734774887561798 ] }, { "rounds": 4, "changes": [ 0.04730474203824997, 0.04924433305859566, 0.04874498397111893 ] } ] }