Robust Physics-Sparse Neural Dynamics / report_bench_2026-08-31T193914.md

Failed on benchmark

Raw ⬇ ZIP

Стенд-проверка (stage-2) · промт оператора:

(универсальный)

Ответ агента:

{ "worked": false, "confidence": 10, "verdict": "The required stage-2 bench was run on the registered built-in dynamics track using 8 paired seeds, a tuned baseline sweep, and the same learning-rate union for both systems. The robust physics-sparse hybrid was significantly worse than the standard rnn_small baseline, with paired delta +0.03931 and permutation p=0.0081; the trained-model stability signature also failed to confirm the predicted reduction in amplification.", "metrics": { "baseline": "Full 8-seed test MSE 0.0005546554; tuned lr=0.01.", "idea": "Best 8-seed test MSE 0.03986520 at lr=0.01; paired delta +0.03931054; permutation p=0.0081; idea worse significantly." }, "bench_report": { "bench_version": 1, "track": "dynamics", "model": "rnn_small", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": { "lr": 0.01 }, "sweep": [ { "cfg": { "lr": 0.001 }, "mean": 0.001921573159052059 }, { "cfg": { "lr": 0.003 }, "mean": 0.0011179699067724869 }, { "cfg": { "lr": 0.01 }, "mean": 0.0006299250526353717 } ], "full": { "mean": 0.0005546553857129766, "std": 0.00021396768993230646, "per_seed": [ 0.0004971042508259416, 0.0004999673110432923, 0.00046853424282744527, 0.0010540944058448076, 0.0005468997405841947, 0.00023900401720311493, 0.0006055250996723771, 0.0005261140177026391 ], "n": 8 } }, "idea": { "mean": 0.03986519586760551, "std": 0.034170930267310405, "per_seed": [ 0.028408890590071678, 0.1224212571978569, 0.01613774709403515, 0.053532518446445465, 0.046550970524549484, 0.02247580699622631, 0.018617291003465652, 0.010777085088193417 ], "n": 8 }, "comparison": { "delta_mean": 0.03931054048189253, "idea_wins": 0, "n_pairs": 8, "per_seed_diffs": [ 0.027911786339245737, 0.12192128988681361, 0.015669212851207703, 0.05247842404060066, 0.04600407078396529, 0.022236802979023196, 0.018011765903793275, 0.010250971070490777 ], "p_value": 0.0081, "mde": 0.030512142347341273, "mde_rel_pct": 5501.099084816372, "verdict": "idea worse (significant)", "system_worked": false }, "mechanism_signature": { "nn_scale_stability": { "prediction": "known stable relaxation/readout reduces local rollout amplification", "baseline": { "mean_jacobian_norm": 0.3944494118914008, "max_jacobian_norm": 0.4280492067337036, "perturbation_proxy": 0.06204786151647568 }, "idea": { "mean_jacobian_norm": 0.9194773212075233, "max_jacobian_norm": 13.631381034851074, "perturbation_proxy": 0.8245298266410828 }, "confirmed": false }, "idea_cfg": { "lr": 0.01 } }, "protocol_note": "8 paired seeds; 400/160 samples; 18 epochs; baseline and idea evaluated on identical lr union." }, "how_to_run": "/home/maxwelhelp/main/bin/python3 bench_experiment.py", "files": [ "bench_experiment.py", "bench_report.json" ], "limitations": "Only the registered built-in actuated-pendulum supervised finite-horizon prediction task was tested, not the full alternating encoder/TLS-RANSAC derivative-identification procedure, corrupted-observation robustness, long-horizon rollout boundary sweep, or external physical data.", "system_verdict": "failed", "practical_verdict": "harms", "mechanism_ok": 0, "system_judged": true }