{ "bench_version": 1, "track": "sequence", "model": "transformer_tiny", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": { "lr": 0.001, "rank": 32 }, "sweep": [ { "cfg": { "lr": 0.001, "rank": 32 }, "mean": 0.3797149360179901 }, { "cfg": { "lr": 0.003, "rank": 32 }, "mean": 0.5440608412027359 }, { "cfg": { "lr": 0.005, "rank": 32 }, "mean": 0.6877395510673523 } ], "full": { "mean": 0.3990439400076866, "std": 0.03555091686816573, "per_seed": [ 0.40484797954559326, 0.35254013538360596, 0.3537353277206421, 0.40773630142211914, 0.3955448567867279, 0.39521750807762146, 0.4743589758872986, 0.4083704352378845 ], "n": 8 } }, "idea": { "mean": 0.4859609939157963, "std": 0.030690392503688, "per_seed": [ 0.5398046374320984, 0.4721372127532959, 0.4599805176258087, 0.4985331892967224, 0.46909099817276, 0.45167359709739685, 0.5286514163017273, 0.46781638264656067 ], "n": 8 }, "comparison": { "delta_mean": 0.08691705390810966, "idea_wins": 0, "n_pairs": 8, "per_seed_diffs": [ 0.13495665788650513, 0.11959707736968994, 0.10624518990516663, 0.09079688787460327, 0.0735461413860321, 0.05645608901977539, 0.05429244041442871, 0.05944594740867615 ], "p_value": 0.0081, "mde": 0.025849569428390114, "mde_rel_pct": 6.477875450982211, "verdict": "idea worse (significant)", "system_worked": false }, "idea_sweep": [ { "cfg": { "lr": 0.001, "rank": 12 }, "result": { "mean": 0.8132649660110474, "std": 0.07039976968716767, "per_seed": [ 0.8669847846031189, 0.6813386082649231, 0.7922707200050354, 0.7799481153488159, 0.8783856630325317, 0.7520660758018494, 0.8496934175491333, 0.9054323434829712 ], "n": 8 } }, { "cfg": { "lr": 0.001, "rank": 16 }, "result": { "mean": 0.6447810083627701, "std": 0.0902293737866928, "per_seed": [ 0.8473581075668335, 0.532031774520874, 0.5603737235069275, 0.6733526587486267, 0.6476457715034485, 0.6398608684539795, 0.6668751835823059, 0.590749979019165 ], "n": 8 } }, { "cfg": { "lr": 0.001, "rank": 20 }, "result": { "mean": 0.4859609939157963, "std": 0.030690392503688, "per_seed": [ 0.5398046374320984, 0.4721372127532959, 0.4599805176258087, 0.4985331892967224, 0.46909099817276, 0.45167359709739685, 0.5286514163017273, 0.46781638264656067 ], "n": 8 } }, { "cfg": { "lr": 0.003, "rank": 12 }, "result": { "mean": 0.8662505745887756, "std": 0.08758038922827249, "per_seed": [ 0.9283100366592407, 0.69541335105896, 0.9051690101623535, 0.8301766514778137, 0.9451277852058411, 0.7725451588630676, 0.9667305946350098, 0.8865320086479187 ], "n": 8 } }, { "cfg": { "lr": 0.003, "rank": 16 }, "result": { "mean": 0.6834343373775482, "std": 0.09683352283532777, "per_seed": [ 0.8036410808563232, 0.5426891446113586, 0.5894339084625244, 0.8408225774765015, 0.6754513382911682, 0.6724597215652466, 0.7292736768722534, 0.6137032508850098 ], "n": 8 } }, { "cfg": { "lr": 0.003, "rank": 20 }, "result": { "mean": 0.5512603707611561, "std": 0.09140216767338745, "per_seed": [ 0.7226142287254333, 0.4266541600227356, 0.5001299381256104, 0.5239855051040649, 0.459748238325119, 0.5338931083679199, 0.6257912516593933, 0.6172665357589722 ], "n": 8 } }, { "cfg": { "lr": 0.005, "rank": 12 }, "result": { "mean": 0.9177583232522011, "std": 0.11157952187308297, "per_seed": [ 1.0815995931625366, 0.7145638465881348, 0.9597082734107971, 0.8518447875976562, 1.0344592332839966, 0.8185186982154846, 0.9612360000610352, 0.9201361536979675 ], "n": 8 } }, { "cfg": { "lr": 0.005, "rank": 16 }, "result": { "mean": 0.768573522567749, "std": 0.1272932016139664, "per_seed": [ 0.9035587906837463, 0.5587367415428162, 0.6324987411499023, 0.8873728513717651, 0.7445271611213684, 0.9305623769760132, 0.804309606552124, 0.6870219111442566 ], "n": 8 } }, { "cfg": { "lr": 0.005, "rank": 20 }, "result": { "mean": 0.5918549560010433, "std": 0.11067653453007016, "per_seed": [ 0.8396977186203003, 0.4893060028553009, 0.5174769163131714, 0.6687707304954529, 0.5009192824363708, 0.5292778015136719, 0.6293551921844482, 0.5600360035896301 ], "n": 8 } } ], "audit": { "structural_match": "multi-token sequence forecast", "theorem_rank_asymptotic": 15.311856320206266, "empirical_rank": 15, "chosen_rank_grid": [ 12, 16, 20 ], "epochs": 8, "n_train": 800, "n_test": 300 }, "mechanism_signature": { "prediction": "DPSS reconstruction error < Fourier at theorem rank", "predicted_dpss_lt_fourier": true, "observed_dpss_error_mean": 0.21955212997938756, "observed_fourier_error_mean": 0.20034139348637556, "observed_relative_improvement": -0.0958900013557031, "attention_work_fraction": 0.25, "trained_model_samples": [ { "seed": 0, "dpss_reconstruction_error": 0.22673050321714594, "fourier_reconstruction_error": 0.20555503007570636, "prediction_rms": 0.6146740913391113 }, { "seed": 1, "dpss_reconstruction_error": 0.221470346993945, "fourier_reconstruction_error": 0.1998581531341289, "prediction_rms": 0.6230881810188293 }, { "seed": 2, "dpss_reconstruction_error": 0.19929562370165232, "fourier_reconstruction_error": 0.18154581981703843, "prediction_rms": 0.6864827871322632 }, { "seed": 3, "dpss_reconstruction_error": 0.2255786421582673, "fourier_reconstruction_error": 0.210578079503818, "prediction_rms": 0.7499093413352966 }, { "seed": 4, "dpss_reconstruction_error": 0.2059191288393489, "fourier_reconstruction_error": 0.1945450930599946, "prediction_rms": 0.6713075637817383 }, { "seed": 5, "dpss_reconstruction_error": 0.21693499314985126, "fourier_reconstruction_error": 0.19918181458416848, "prediction_rms": 0.580341637134552 }, { "seed": 6, "dpss_reconstruction_error": 0.2512841631045155, "fourier_reconstruction_error": 0.22277636588465038, "prediction_rms": 0.654333770275116 }, { "seed": 7, "dpss_reconstruction_error": 0.2092036386703743, "fourier_reconstruction_error": 0.1886907918314992, "prediction_rms": 0.7088998556137085 } ], "confirmed": false } }