{ "bench_version": 1, "track": "dynamics", "model": "rnn_small", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": { "lr": 0.006 }, "sweep": [ { "cfg": { "lr": 0.001 }, "mean": 0.5088173821568489 }, { "cfg": { "lr": 0.003 }, "mean": 0.06073589529842138 }, { "cfg": { "lr": 0.006 }, "mean": 0.023900731466710567 } ], "full": { "mean": 0.019088300527073443, "std": 0.00558386015869873, "per_seed": [ 0.018445175141096115, 0.023440849035978317, 0.02936750277876854, 0.024349398910999298, 0.01488791685551405, 0.012780136428773403, 0.01545807532966137, 0.013977349735796452 ], "n": 8 } }, "idea": { "best_cfg": { "lr": 0.006 }, "sweep": [ { "cfg": { "lr": 0.001 }, "mean": 0.3442936111241579, "std": 0.11055745967503162, "per_seed": [ 0.4150988757610321, 0.572791337966919, 0.3577822744846344, 0.3750651478767395, 0.2974511682987213, 0.17869944870471954, 0.254057914018631, 0.30340272188186646 ], "n": 8 }, { "cfg": { "lr": 0.003 }, "mean": 0.03436855808831751, "std": 0.01405565796915205, "per_seed": [ 0.05125405639410019, 0.041576966643333435, 0.04229721054434776, 0.055242300033569336, 0.021142112091183662, 0.016799848526716232, 0.020402031019330025, 0.026233939453959465 ], "n": 8 }, { "cfg": { "lr": 0.006 }, "mean": 0.015928541193716228, "std": 0.004679734568777004, "per_seed": [ 0.0178981926292181, 0.013649515807628632, 0.024342793971300125, 0.021126413717865944, 0.012977873906493187, 0.008715411648154259, 0.01320882048457861, 0.015509307384490967 ], "n": 8 } ], "mean": 0.015928541193716228, "std": 0.004679734568777004, "per_seed": [ 0.0178981926292181, 0.013649515807628632, 0.024342793971300125, 0.021126413717865944, 0.012977873906493187, 0.008715411648154259, 0.01320882048457861, 0.015509307384490967 ], "n": 8 }, "comparison": { "delta_mean": -0.003159759333357215, "idea_wins": 7, "n_pairs": 8, "per_seed_diffs": [ -0.0005469825118780136, -0.009791333228349686, -0.005024708807468414, -0.003222985193133354, -0.0019100429490208626, -0.0040647247806191444, -0.00224925484508276, 0.0015319576486945152 ], "p_value": 0.02315, "mde": 0.002818017003243901, "mde_rel_pct": 14.763058655992097, "verdict": "idea better (significant)", "system_worked": true }, "mechanism_signature": { "collective_detectability": { "window": 2, "predicted_lambda_min_positive": true, "observed_mean_fused_information": 1.459951639175415, "observed_positive_fraction": 1.0, "tolerance": "positivity and >0.99 positive fraction", "confirmed": true } }, "protocol_notes": { "structural_match": "dynamics: recurrent pendulum rollout and latent-state stability", "paired_seeds": 8, "epochs": 12, "batch": 128, "shared_lr_union": [ 0.001, 0.003, 0.006 ], "baseline_method_knob": "equal arithmetic mean (fixed by definition)", "system_parity": "separately trained identical two-agent GRU encoders and predictor; only fusion differs" } }