# Стенд-проверка (stage-2) · промт оператора: (универсальный) Ответ агента: { "worked": true, "confidence": 8, "verdict": "Natural-parameter information fusion beat the matched arithmetic-fusion recurrent system on the registered dynamics track. Mean test MSE was 0.0159285412 versus 0.0190883005, with 7/8 paired wins and permutation p=0.02315, satisfying the significant-win criterion.", "metrics": { "baseline": "mean MSE 0.0190883005; std 0.0055838602; best lr 0.006", "idea": "mean MSE 0.0159285412; std 0.0046797346; best lr 0.006", "paired_delta": "-0.0031597593", "p_value": "0.02315", "wins": "7/8" }, "how_to_run": "/home/maxwelhelp/main/bin/python3 bench_experiment.py", "files": [ "bench_experiment.py", "bench_report.json" ], "limitations": "Only the registered dynamics track was tested. The implementation uses learned scalar precisions and fixed complementary masks rather than explicit autodifferentiated Jacobians and noise matrices. Stale-message schedules, graph sweeps, longer rollouts, and quantitative full-Gramian eigenvalue estimation were not tested.", "bench_report": { "bench_version": 1, "track": "dynamics", "model": "rnn_small", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": { "lr": 0.006 }, "sweep": [ { "cfg": { "lr": 0.001 }, "mean": 0.5088173821568489 }, { "cfg": { "lr": 0.003 }, "mean": 0.06073589529842138 }, { "cfg": { "lr": 0.006 }, "mean": 0.023900731466710567 } ], "full": { "mean": 0.019088300527073443, "std": 0.00558386015869873, "per_seed": [ 0.018445175141096115, 0.023440849035978317, 0.02936750277876854, 0.024349398910999298, 0.01488791685551405, 0.012780136428773403, 0.01545807532966137, 0.013977349735796452 ], "n": 8 } }, "idea": { "best_cfg": { "lr": 0.006 }, "sweep": [ { "cfg": { "lr": 0.001 }, "mean": 0.3442936111241579 }, { "cfg": { "lr": 0.003 }, "mean": 0.03436855808831751 }, { "cfg": { "lr": 0.006 }, "mean": 0.015928541193716228 } ], "mean": 0.015928541193716228, "std": 0.004679734568777004, "per_seed": [ 0.0178981926292181, 0.013649515807628632, 0.024342793971300125, 0.021126413717865944, 0.012977873906493187, 0.008715411648154259, 0.01320882048457861, 0.015509307384490967 ], "n": 8 }, "comparison": { "delta_mean": -0.003159759333357215, "idea_wins": 7, "n_pairs": 8, "per_seed_diffs": [ -0.0005469825118780136, -0.009791333228349686, -0.005024708807468414, -0.003222985193133354, -0.0019100429490208626, -0.0040647247806191444, -0.00224925484508276, 0.0015319576486945152 ], "p_value": 0.02315, "verdict": "idea better (significant)", "system_worked": true }, "mechanism_signature": { "collective_detectability": { "window": 2, "predicted_lambda_min_positive": true, "observed_mean_fused_information": 1.459951639175415, "observed_positive_fraction": 1.0, "tolerance": "positivity and >0.99 positive fraction", "confirmed": true } } }, "system_verdict": "worked", "practical_verdict": "helps", "mechanism_ok": 0, "system_judged": true }