# Стенд-проверка (stage-2) · промт оператора: (универсальный) Ответ агента: { "worked": false, "confidence": 9, "verdict": "The registered built-in dynamics track was rerun with 8 paired seeds, a tuned three-point baseline sweep, and the same learning-rate grid for the idea. The matching-controllable recurrent model had higher test MSE than the dense baseline (0.004187 versus 0.003113), paired delta +0.001074, and permutation p=0.3437; therefore there was no significant win. The trained-model mechanism signature was also not confirmed: numerical controllability ranks were 15–18/24 rather than full rank.", "metrics": { "baseline": "Best lr=0.01, weight_decay=0.0; full 8-seed test MSE mean 0.0031130631 ± 0.0015298698.", "idea": "Best lr=0.01, weight_decay=0.0; full 8-seed test MSE mean 0.0041867752 ± 0.0021045838.", "delta_mean": 0.0010737121192505583, "p_value": 0.3437 }, "bench_report": { "bench_version": 1, "track": "dynamics", "model": "explicit_tanh_rnn_shared", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": { "lr": 0.01, "weight_decay": 0.0 }, "sweep": [ { "cfg": { "lr": 0.001, "weight_decay": 0.0 }, "mean": 0.32027048990130424 }, { "cfg": { "lr": 0.003, "weight_decay": 0.0 }, "mean": 0.01070007259841077 }, { "cfg": { "lr": 0.01, "weight_decay": 0.0 }, "mean": 0.0029869703284930438 } ], "full": { "mean": 0.0031130630668485537, "std": 0.0015298698474074236, "per_seed": [ 0.002583631779998541, 0.001341367606073618, 0.0019145020050927997, 0.006108379922807217, 0.005108166951686144, 0.002453721361234784, 0.0029577165842056274, 0.002437018323689699 ], "n": 8 } }, "idea": { "mean": 0.004186775186099112, "std": 0.0021045838143091996, "per_seed": [ 0.0009666543919593096, 0.005662341136485338, 0.0049017248675227165, 0.0013687764294445515, 0.007184116635471582, 0.0038662750739604235, 0.003245178610086441, 0.006299134343862534 ], "n": 8 }, "comparison": { "delta_mean": 0.0010737121192505583, "idea_wins": 2, "n_pairs": 8, "per_seed_diffs": [ -0.0016169773880392313, 0.00432097353041172, 0.002987222862429917, -0.004739603493362665, 0.0020759496837854385, 0.0014125537127256393, 0.0002874620258808136, 0.0038621160201728344 ], "p_value": 0.3437, "mde": 0.002543128344355023, "mde_rel_pct": 81.69215623792383, "verdict": "no significant win", "system_worked": false }, "mechanism_signature": { "prediction": "row-saturating matching gives full-rank trained controllability and improves long-horizon input access", "trained_model_measurements": { "ranks": [ 16, 17, 17, 15, 15, 18, 17, 16 ], "min_singular_range": [ 1.5868783161841502e-24, 6.442173137209217e-16 ], "max_singular_range": [ 0.7894802689552307, 1.0363144874572754 ] }, "confirmed": false } }, "how_to_run": "/home/maxwelhelp/main/bin/python3 matching_bench.py", "files": [ "matching_bench.py", "bench_report.json", "bench_run.log", "bench_run_rerun.log" ], "limitations": "Only the registered built-in dynamics pendulum track was tested. No multi-context Gramian penalty, alternative masks, larger hidden sizes, or longer training budgets were evaluated; numerical rank is tolerance-dependent.", "system_verdict": "failed", "practical_verdict": "inconclusive", "mechanism_ok": 0, "system_judged": true }