Phase-Aware Bias-Energy Trust Region / report_bench_2026-09-01T172236.md
Beats tuned baseline
Стенд-проверка (stage-2) · промт оператора:
(универсальный)
Ответ агента:
{ "worked": true, "confidence": 7, "verdict": "The phase-aware controller was implemented as a training-loop modification using the shared mlp_tiny model on the registered tabular optimizer track. It beat tuned fixed clipping on mean test MSE, 7 of 8 paired seeds, with delta_mean=-0.67497 and permutation p=0.04815, satisfying the benchmark's significant-win criterion. The mechanism signature was not confirmed because the energy target was not reached and tau saturated at 10.0.", "metrics": { "baseline": "Fixed clipping best_cfg={lr:0.006,tau:0.5}, full mean MSE=8.051172316074371", "idea": "Phase-aware best config={lr:0.006,tau0:0.5,eta:0.09}, mean MSE=7.376198410987854, delta=-0.6749739050865173, p=0.04815" }, "how_to_run": "/home/maxwelhelp/main/bin/python3 stage2_bench.py", "files": [ "stage2_bench.py", "bench_report.json", "run_stage2.txt" ], "bench_report": { "bench_version": 1, "track": "tabular", "model": "mlp_tiny", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": { "lr": 0.006, "tau": 0.5 }, "sweep": [ { "cfg": { "lr": 0.001, "tau": 0.5 }, "mean": 50.39839839935303 }, { "cfg": { "lr": 0.001, "tau": 1.0 }, "mean": 50.397457122802734 }, { "cfg": { "lr": 0.001, "tau": 2.0 }, "mean": 50.39675712585449 }, { "cfg": { "lr": 0.002, "tau": 0.5 }, "mean": 14.087330341339111 }, { "cfg": { "lr": 0.002, "tau": 1.0 }, "mean": 14.09030532836914 }, { "cfg": { "lr": 0.002, "tau": 2.0 }, "mean": 14.094738960266113 }, { "cfg": { "lr": 0.003, "tau": 0.5 }, "mean": 10.800803184509277 }, { "cfg": { "lr": 0.003, "tau": 1.0 }, "mean": 10.79708456993103 }, { "cfg": { "lr": 0.003, "tau": 2.0 }, "mean": 10.818211793899536 }, { "cfg": { "lr": 0.004, "tau": 0.5 }, "mean": 8.752828359603882 }, { "cfg": { "lr": 0.004, "tau": 1.0 }, "mean": 8.755794763565063 }, { "cfg": { "lr": 0.004, "tau": 2.0 }, "mean": 8.77496600151062 }, { "cfg": { "lr": 0.006, "tau": 0.5 }, "mean": 7.511728048324585 }, { "cfg": { "lr": 0.006, "tau": 1.0 }, "mean": 7.649782061576843 }, { "cfg": { "lr": 0.006, "tau": 2.0 }, "mean": 7.626705288887024 } ], "full": { "mean": 8.051172316074371, "std": 1.1549823103901204, "per_seed": [ 7.279676914215088, 6.340634822845459, 7.740776062011719, 8.685824394226074, 9.63039493560791, 6.726284503936768, 8.459394454956055, 9.546392440795898 ], "n": 8 } }, "idea": { "config": { "lr": 0.006, "tau0": 0.5, "eta": 0.09 }, "mean": 7.376198410987854, "std": 0.8078643252957813, "per_seed": [ 6.912933349609375, 5.868978977203369, 8.128691673278809, 8.1602783203125, 7.357755661010742, 6.566903114318848, 8.233548164367676, 7.780498027801514 ], "n": 8, "diagnostics": [ { "metric": 6.912933349609375, "update_norm_mean": 18.64707014958064, "clip_fraction": 0.62500000724362, "residual": 15.457989124257843, "energy": 2.6955993475714886, "final_tau": 10.0, "phase": "energy" }, { "metric": 5.868978977203369, "update_norm_mean": 15.81644619256258, "clip_fraction": 0.5625000053809749, "residual": 13.15220600542509, "energy": 2.135818994000646, "final_tau": 10.0, "phase": "energy" }, { "metric": 8.128691673278809, "update_norm_mean": 14.61398777945174, "clip_fraction": 0.5555555599017276, "residual": 12.13080305696672, "energy": 1.9603608226717673, "final_tau": 10.0, "phase": "energy" }, { "metric": 8.1602783203125, "update_norm_mean": 18.281507012744743, "clip_fraction": 0.6157407452248864, "residual": 15.370208648343882, "energy": 2.409127195988459, "final_tau": 10.0, "phase": "energy" }, { "metric": 7.357755661010742, "update_norm_mean": 16.17437604152494, "clip_fraction": 0.5902777841935555, "residual": 13.253949597995314, "energy": 2.4005918734516047, "final_tau": 10.0, "phase": "energy" }, { "metric": 6.566903114318848, "update_norm_mean": 19.33727982143561, "clip_fraction": 0.650462968274951, "residual": 16.040926492462557, "energy": 2.8265877289300914, "final_tau": 10.0, "phase": "energy" }, { "metric": 8.233548164367676, "update_norm_mean": 18.142026902900803, "clip_fraction": 0.5925925980425544, "residual": 15.166145478220036, "energy": 2.4767839011183406, "final_tau": 10.0, "phase": "energy" }, { "metric": 7.780498027801514, "update_norm_mean": 16.652106201483143, "clip_fraction": 0.5717592653301027, "residual": 13.809878197809061, "energy": 2.3481363373500703, "final_tau": 10.0, "phase": "energy" } ] }, "comparison": { "delta_mean": -0.6749739050865173, "idea_wins": 7, "n_pairs": 8, "per_seed_diffs": [ -0.3667435646057129, -0.47165584564208984, 0.38791561126708984, -0.5255460739135742, -2.272639274597168, -0.15938138961791992, -0.2258462905883789, -1.7658944129943848 ], "p_value": 0.04815, "mde": 0.7410767906092078, "mde_rel_pct": 9.204582407578448, "verdict": "idea better (significant)", "system_worked": true }, "mechanism_signature": { "prediction": "alpha <= p*beta selects energy control; empirical energy approaches target", "alpha": 1.0, "beta": 1.0, "p": 2.0, "predicted_phase": "energy", "observed_phase": "energy", "observed_energy_mean": 2.4066257751353084, "energy_target": 0.22, "observed_residual_mean": 14.297763325185063, "observed_final_tau_mean": 10.0, "confirmed": false } }, "limitations": "Only the registered tabular/Friedman regression track was tested; vision, sequence, dynamics, Transformer, WikiText-103, synthetic Pareto perturbations, and recovery-time comparisons were not tested. The benchmark used 400 training samples, 200 test samples, and 18 epochs. The mechanism controller saturated its tau clamp and missed its empirical energy target, so the significant task-metric win does not validate the proposed quantitative controller calibration.", "system_verdict": "worked", "practical_verdict": "helps", "mechanism_ok": 0, "system_judged": true }