Causal E/I Micro-Event Cell / report_bench_2026-09-01T161842.md
Mechanism confirmed, baseline not beaten
Стенд-проверка (stage-2) · промт оператора:
(универсальный)
Ответ агента:
{ "worked": false, "confidence": 9, "verdict": "The causal E/I micro-event cell was implemented as a matched recurrent dynamics system and evaluated end-to-end against signed macro-step aggregation. The mechanism signature was confirmed through nonzero trained-model behavioral differences, but test MSE was significantly worse for the idea: 0.03774771 versus 0.03009093, paired delta +0.00765679, permutation p=0.0081; therefore there is no benchmark win.", "metrics": { "baseline": "best_cfg={lr:0.003,epochs:12}; sweep means lr=0.0015: 0.03892234, lr=0.003: 0.03009093, lr=0.006: 0.05556619; full mean=0.03009093, std=0.00407143, per_seed=[0.03666029,0.02726038,0.03030632,0.03209355,0.02480439,0.03368592,0.02425563,0.03166095]", "idea": "best_cfg={lr:0.003,epochs:12}; sweep means lr=0.0015: 0.05525698, lr=0.003: 0.03774771, lr=0.006: 0.04161697; mean=0.03774771, std=0.00409269, per_seed=[0.04495187,0.03703208,0.03814383,0.04309968,0.03429987,0.03708112,0.03175383,0.03561943]" }, "bench_report": { "bench_version": 1, "track": "dynamics", "model": "rnn_small", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": { "lr": 0.003, "epochs": 12 }, "sweep": [ { "cfg": { "lr": 0.0015, "epochs": 12 }, "mean": 0.03892233967781067 }, { "cfg": { "lr": 0.003, "epochs": 12 }, "mean": 0.030090928077697754 }, { "cfg": { "lr": 0.006, "epochs": 12 }, "mean": 0.05556618981063366 } ], "full": { "mean": 0.030090928077697754, "std": 0.004071433001350898, "per_seed": [ 0.03666028752923012, 0.02726038359105587, 0.030306316912174225, 0.03209355100989342, 0.024804387241601944, 0.033685918897390366, 0.024255627766251564, 0.03166095167398453 ], "n": 8 } }, "idea": { "mean": 0.03774771373718977, "std": 0.004092688612850228, "per_seed": [ 0.044951874762773514, 0.037032078951597214, 0.0381438285112381, 0.04309968277812004, 0.03429986909031868, 0.037081118673086166, 0.03175382688641548, 0.035619430243968964 ], "n": 8, "best_config": { "lr": 0.003, "epochs": 12 } }, "comparison": { "delta_mean": 0.007656785659492016, "idea_wins": 0, "n_pairs": 8, "per_seed_diffs": [ 0.008291587233543396, 0.009771695360541344, 0.007837511599063873, 0.011006131768226624, 0.009495481848716736, 0.0033951997756958008, 0.0074981991201639175, 0.003958478569984436 ], "p_value": 0.0081, "mde": 0.0022651487884256674, "mde_rel_pct": 7.527680045550038, "verdict": "idea worse (significant)", "system_worked": false }, "mechanism_signature": { "predicted": { "unsafe_fraction": "elevated near mixed signed threshold", "order_effect": "nonzero" }, "observed": { "high_drive_fraction": 0.25, "mean_output_abs_difference": 0.15396859403699636 }, "confirmed": true }, "protocol": { "seeds": [ 0, 1, 2, 3, 4, 5, 6, 7 ], "n_train": 400, "n_test": 400, "matched_architecture": true, "shared_hyperparameter_union": [ { "lr": 0.0015, "epochs": 12 }, { "lr": 0.003, "epochs": 12 }, { "lr": 0.006, "epochs": 12 } ] } }, "how_to_run": "/home/maxwelhelp/main/bin/python3 causal_ei_bench.py", "files": [ "causal_ei_bench.py", "bench_report.json" ], "limitations": "Only the registered built-in dynamics track was tested. No larger datasets, longer training, speed/FLOP measurements, spiking classification track, or alternative timestamp distributions were evaluated. The mechanism signature confirms behavioral divergence but does not imply task improvement.", "system_verdict": "partial", "practical_verdict": "harms", "mechanism_ok": 1, "system_judged": true }