Delay-Aware Frequency-Preserving Recurrent Coupling / report_bench_2026-08-31T123143.md
Mechanism confirmed, baseline not beaten
Стенд-проверка (stage-2) · промт оператора:
(универсальный)
Ответ агента:
{ "worked": false, "confidence": 9, "verdict": "The delay-aware compensated coupled GRU was evaluated on the registered dynamics/control track with the required paired-seed protocol. The mechanism signature was confirmed behaviorally, but the best idea setting selected zero coupling and exactly matched the tuned baseline; nonzero compensated coupling was worse. Therefore the bench verdict is no measurable effect and worked=false.", "metrics": { "baseline": "best_cfg={lr:0.006,coupling:0.0}; full mean test MSE=0.002108921129547525, std=0.001215784166403164, per_seed=[0.0009661344229243696,0.0021135066635906696,0.0017933575436472893,0.0040876418352127075,0.0011596226831898093,0.0011802533408626914,0.004168401472270489,0.0014024510746821761]", "idea": "best_cfg={lr:0.006,coupling:0.0}; mean test MSE=0.002108921129547525, std=0.001215784166403164, per_seed=[0.0009661344229243696,0.0021135066635906696,0.0017933575436472893,0.0040876418352127075,0.0011596226831898093,0.0011802533408626914,0.004168401472270489,0.0014024510746821761]; compensated lr=0.001,coupling=0.08 mean=0.046488794789183885; compensated lr=0.006,coupling=0.16 mean=0.0030070416687522084" }, "bench_report": { "bench_version": 1, "track": "dynamics", "model": "rnn_small", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": { "lr": 0.006, "coupling": 0.0 }, "sweep": [ { "cfg": { "lr": 0.001, "coupling": 0.0 }, "mean": 0.03594269044697285 }, { "cfg": { "lr": 0.001, "coupling": 0.08 }, "mean": 0.03594269044697285 }, { "cfg": { "lr": 0.001, "coupling": 0.16 }, "mean": 0.03594269044697285 }, { "cfg": { "lr": 0.003, "coupling": 0.0 }, "mean": 0.003497575788060203 }, { "cfg": { "lr": 0.003, "coupling": 0.08 }, "mean": 0.003497575788060203 }, { "cfg": { "lr": 0.003, "coupling": 0.16 }, "mean": 0.003497575788060203 }, { "cfg": { "lr": 0.006, "coupling": 0.0 }, "mean": 0.002240160116343759 }, { "cfg": { "lr": 0.006, "coupling": 0.08 }, "mean": 0.002240160116343759 }, { "cfg": { "lr": 0.006, "coupling": 0.16 }, "mean": 0.002240160116343759 } ], "full": { "mean": 0.002108921129547525, "std": 0.001215784166403164, "per_seed": [ 0.0009661344229243696, 0.0021135066635906696, 0.0017933575436472893, 0.0040876418352127075, 0.0011596226831898093, 0.0011802533408626914, 0.004168401472270489, 0.0014024510746821761 ], "n": 8 } }, "idea": { "mean": 0.002108921129547525, "std": 0.001215784166403164, "per_seed": [ 0.0009661344229243696, 0.0021135066635906696, 0.0017933575436472893, 0.0040876418352127075, 0.0011596226831898093, 0.0011802533408626914, 0.004168401472270489, 0.0014024510746821761 ], "n": 8 }, "comparison": { "delta_mean": 0.0, "idea_wins": 0, "n_pairs": 8, "per_seed_diffs": [ 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0, 0.0 ], "p_value": 1.0, "mde": 0.0, "mde_rel_pct": 0.0, "verdict": "no measurable effect", "system_worked": false }, "mechanism_signature": { "trained_model": true, "delay_steps": 1, "baseline_nonzero_state_rms": 0.3115456700325012, "idea_nonzero_state_rms": 0.26382508873939514, "baseline_consensus_state_rms": 0.3213949203491211, "idea_consensus_state_rms": 0.3509313464164734, "observed_delayed_cross_branch_correlation": 0.8421738798093602, "predicted_delayed_phase_alignment": "positive correlation", "confirmed": true }, "idea_sweep": [ { "cfg": { "lr": 0.006, "coupling": 0.0 }, "result": { "mean": 0.002108921129547525, "std": 0.001215784166403164, "per_seed": [ 0.0009661344229243696, 0.0021135066635906696, 0.0017933575436472893, 0.0040876418352127075, 0.0011596226831898093, 0.0011802533408626914, 0.004168401472270489, 0.0014024510746821761 ], "n": 8 } }, { "cfg": { "lr": 0.001, "coupling": 0.08 }, "result": { "mean": 0.046488794789183885, "std": 0.04397113823007572, "per_seed": [ 0.003001284785568714, 0.15231122076511383, 0.03723231330513954, 0.03665262460708618, 0.004484621342271566, 0.0410233773291111, 0.06261226534843445, 0.0345926508307457 ], "n": 8 } }, { "cfg": { "lr": 0.006, "coupling": 0.16 }, "result": { "mean": 0.0030070416687522084, "std": 0.0014159174184431745, "per_seed": [ 0.0014849627623334527, 0.002292004879564047, 0.003297199495136738, 0.006229760590940714, 0.001956953899934888, 0.0019148519495502114, 0.0036266844253987074, 0.003253915347158909 ], "n": 8 } } ], "protocol": { "paired_seeds": [ 0, 1, 2, 3, 4, 5, 6, 7 ], "epochs": 12, "batch": 128, "structural_match": "dynamics/control", "baseline_grid": [ { "lr": 0.001, "coupling": 0.0 }, { "lr": 0.001, "coupling": 0.08 }, { "lr": 0.001, "coupling": 0.16 }, { "lr": 0.003, "coupling": 0.0 }, { "lr": 0.003, "coupling": 0.08 }, { "lr": 0.003, "coupling": 0.16 }, { "lr": 0.006, "coupling": 0.0 }, { "lr": 0.006, "coupling": 0.08 }, { "lr": 0.006, "coupling": 0.16 } ] } }, "how_to_run": "/home/maxwelhelp/main/bin/python3 delay_bench.py", "files": [ "delay_bench.py", "bench_report.json" ], "limitations": "Only the registered built-in dynamics track was tested. The experiment used a one-step delay, short 12-epoch training, 400 training and 200 test examples, and did not test longer horizons, multiple delays, broader oscillatory frequency tasks, larger recurrent architectures, or wall-clock/FLOP effects. The behavioral signature confirms delayed cross-branch alignment in trained hidden states but not preservation of a known target frequency on the benchmark task.", "system_verdict": "partial", "practical_verdict": "no_effect", "mechanism_ok": 1, "system_judged": true }