# Стенд-проверка (stage-2) · промт оператора: (универсальный) Ответ агента: { "worked": false, "confidence": 10, "verdict": "The feasibility-preserving compensator was implemented as a matched rnn_small system on the registered dynamics track. Its mechanism signature was confirmed: predicted spectral decay 0.88227 versus observed 0.88158, with anti-windup peak magnitude 0.16032 versus ordinary integral 0.70819. However, it significantly lost on rollout MSE, so the idea did not transfer successfully to neural benchmark training.", "metrics": { "baseline": "mean MSE 0.0014198302378645167; best lr 0.01, epochs 12", "idea": "mean MSE 0.47640712931752205; best KI 0.05, KA 0.7, lr 0.01, epochs 12", "comparison": "delta_mean +0.47498729907965753; permutation p=0.0081; idea wins 0/8" }, "how_to_run": "/home/maxwelhelp/main/bin/python3 bench_experiment.py", "files": [ "bench_experiment.py", "bench_report.json", "math_check.json" ], "limitations": "Only the registered dynamics track and rnn_small model were tested; no noisy measurements, multidimensional task space, explicit actuator constraint suite, or longer-horizon deployment was evaluated.", "bench_report": { "bench_version": 1, "track": "dynamics", "model": "rnn_small", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": { "lr": 0.01, "epochs": 12 }, "sweep": [ { "cfg": { "lr": 0.001, "epochs": 12 }, "mean": 0.005020207085181028 }, { "cfg": { "lr": 0.003, "epochs": 12 }, "mean": 0.0029489120497601107 }, { "cfg": { "lr": 0.01, "epochs": 12 }, "mean": 0.0014198302378645167 } ], "full": { "mean": 0.0014198302378645167, "std": 0.00039839114901047844, "per_seed": [ 0.0009327387670055032, 0.0018641706556081772, 0.0010785290505737066, 0.0020708823576569557, 0.0009260809747502208, 0.0014055331703275442, 0.0016443432541564107, 0.001436363672837615 ], "n": 8 } }, "idea": { "best_cfg": { "lr": 0.01, "epochs": 12, "ki": 0.05, "ka": 0.7 }, "mean": 0.47640712931752205, "std": 0.04352936722428276, "per_seed": [ 0.4626535177230835, 0.45927271246910095, 0.5099033713340759, 0.5794274210929871, 0.4526548683643341, 0.45540836453437805, 0.4523857831954956, 0.4395509958267212 ], "n": 8 }, "comparison": { "delta_mean": 0.47498729907965753, "idea_wins": 0, "n_pairs": 8, "per_seed_diffs": [ 0.461720778956078, 0.4574085418134928, 0.5088248422835022, 0.5773565387353301, 0.4517287873895839, 0.4540028313640505, 0.4507414399413392, 0.4381146321538836 ], "p_value": 0.0081, "verdict": "idea worse (significant)", "system_worked": false }, "mechanism_signature": { "math_and_trained_behavior": { "spectral_radius_predicted": 0.8822698000045112, "decay_factor_observed": 0.8815778323329195, "decay_relative_error": 0.00078430393014483, "ordinary_max_compensator": 0.708185009625288, "antiwindup_max_compensator": 0.1603214553812824, "windup_reduction_ratio": 0.22638357661102013, "constraint_violation": 0.0, "prediction_confirmed": true, "trained_nominal_abs_mean": 0.6836502552032471, "trained_shaped_abs_mean": 0.21633578836917877, "trained_residual_abs_mean": 0.2636767625808716, "trained_compensator_abs_mean": 0.22096218168735504, "confirmed": true }, "idea_sweep": [ { "cfg": { "lr": 0.01, "epochs": 12, "ki": 0.05, "ka": 0.7 }, "mean": 0.47640712931752205, "std": 0.04352936722428276, "n": 8 }, { "cfg": { "lr": 0.01, "epochs": 12, "ki": 0.075, "ka": 0.9 }, "mean": 0.4816158339381218, "std": 0.04396653306668558, "n": 8 }, { "cfg": { "lr": 0.01, "epochs": 12, "ki": 0.1, "ka": 1.1 }, "mean": 0.48671987280249596, "std": 0.044320348769445794, "n": 8 } ] } }, "system_verdict": "failed", "practical_verdict": "harms", "mechanism_ok": 0, "system_judged": true }