{ "bench_version": 1, "track": "sequence", "model": "transformer_tiny", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": { "lr": 0.0015 }, "sweep": [ { "cfg": { "lr": 0.0015 }, "mean": 0.987314760684967 }, { "cfg": { "lr": 0.003 }, "mean": 1.5178270637989044 }, { "cfg": { "lr": 0.006 }, "mean": 2.574708193540573 } ], "full": { "mean": 1.0789237320423126, "std": 0.1999035176851697, "per_seed": [ 0.6216103434562683, 1.084313154220581, 0.9683473706245422, 1.2749881744384766, 1.0933938026428223, 1.1413570642471313, 1.1405138969421387, 1.3068660497665405 ], "n": 8 } }, "idea": { "mean": 1.1184165999293327, "std": 0.17711685282328676, "per_seed": [ 1.2109392881393433, 0.8607760071754456, 0.8473414182662964, 1.2753307819366455, 1.0955066680908203, 1.1688036918640137, 1.0974258184432983, 1.3912091255187988 ], "n": 8, "cfg": { "lr": 0.0015, "beta": 0.5 }, "mechanism_signatures": [ { "state_norm": 0.11540921777486801, "update_norm": 0.6236135959625244, "read_norm": 0.5206019282341003 }, { "state_norm": 0.1203140988945961, "update_norm": 0.7720737457275391, "read_norm": 0.5169996619224548 }, { "state_norm": 0.11553805321455002, "update_norm": 0.6452788710594177, "read_norm": 0.5866501927375793 }, { "state_norm": 0.1047215536236763, "update_norm": 0.48125800490379333, "read_norm": 0.563747763633728 }, { "state_norm": 0.10959173738956451, "update_norm": 0.8530905246734619, "read_norm": 0.4561031460762024 }, { "state_norm": 0.10967692732810974, "update_norm": 0.5639088153839111, "read_norm": 0.5506287217140198 }, { "state_norm": 0.10660543292760849, "update_norm": 0.5371220111846924, "read_norm": 0.5531039237976074 }, { "state_norm": 0.11331523954868317, "update_norm": 0.6734405755996704, "read_norm": 0.5476616621017456 } ] }, "comparison": { "delta_mean": 0.03949286788702011, "idea_wins": 3, "n_pairs": 8, "per_seed_diffs": [ 0.589328944683075, -0.2235371470451355, -0.12100595235824585, 0.0003426074981689453, 0.002112865447998047, 0.027446627616882324, -0.04308807849884033, 0.0843430757522583 ], "p_value": 0.82115, "mde": 0.20215727076091933, "mde_rel_pct": 18.73693800193388, "verdict": "no significant win", "system_worked": false }, "mechanism_signature": { "prediction": "rank-one per-sample state has finite norm and nonzero update/read activity", "observed": [ { "state_norm": 0.11540921777486801, "update_norm": 0.6236135959625244, "read_norm": 0.5206019282341003 }, { "state_norm": 0.1203140988945961, "update_norm": 0.7720737457275391, "read_norm": 0.5169996619224548 }, { "state_norm": 0.11553805321455002, "update_norm": 0.6452788710594177, "read_norm": 0.5866501927375793 }, { "state_norm": 0.1047215536236763, "update_norm": 0.48125800490379333, "read_norm": 0.563747763633728 }, { "state_norm": 0.10959173738956451, "update_norm": 0.8530905246734619, "read_norm": 0.4561031460762024 }, { "state_norm": 0.10967692732810974, "update_norm": 0.5639088153839111, "read_norm": 0.5506287217140198 }, { "state_norm": 0.10660543292760849, "update_norm": 0.5371220111846924, "read_norm": 0.5531039237976074 }, { "state_norm": 0.11331523954868317, "update_norm": 0.6734405755996704, "read_norm": 0.5476616621017456 } ], "idea_sweep": [ { "mean": 1.1184165999293327, "std": 0.17711685282328676, "per_seed": [ 1.2109392881393433, 0.8607760071754456, 0.8473414182662964, 1.2753307819366455, 1.0955066680908203, 1.1688036918640137, 1.0974258184432983, 1.3912091255187988 ], "n": 8, "cfg": { "lr": 0.0015, "beta": 0.5 }, "mechanism_signatures": [ { "state_norm": 0.11540921777486801, "update_norm": 0.6236135959625244, "read_norm": 0.5206019282341003 }, { "state_norm": 0.1203140988945961, "update_norm": 0.7720737457275391, "read_norm": 0.5169996619224548 }, { "state_norm": 0.11553805321455002, "update_norm": 0.6452788710594177, "read_norm": 0.5866501927375793 }, { "state_norm": 0.1047215536236763, "update_norm": 0.48125800490379333, "read_norm": 0.563747763633728 }, { "state_norm": 0.10959173738956451, "update_norm": 0.8530905246734619, "read_norm": 0.4561031460762024 }, { "state_norm": 0.10967692732810974, "update_norm": 0.5639088153839111, "read_norm": 0.5506287217140198 }, { "state_norm": 0.10660543292760849, "update_norm": 0.5371220111846924, "read_norm": 0.5531039237976074 }, { "state_norm": 0.11331523954868317, "update_norm": 0.6734405755996704, "read_norm": 0.5476616621017456 } ] }, { "mean": 1.396845631301403, "std": 0.3391511114620585, "per_seed": [ 0.9106339812278748, 1.3027052879333496, 1.3874930143356323, 0.8688368797302246, 1.8346285820007324, 1.7065383195877075, 1.4406466484069824, 1.7232823371887207 ], "n": 8, "cfg": { "lr": 0.003, "beta": 0.5 }, "mechanism_signatures": [ { "state_norm": 0.12106544524431229, "update_norm": 0.7507100105285645, "read_norm": 0.5045618414878845 }, { "state_norm": 0.10777172446250916, "update_norm": 0.5398299694061279, "read_norm": 0.5512951016426086 }, { "state_norm": 0.13707156479358673, "update_norm": 0.9250611662864685, "read_norm": 0.526276707649231 }, { "state_norm": 0.12658239901065826, "update_norm": 0.7286500334739685, "read_norm": 0.551619827747345 }, { "state_norm": 0.11532067507505417, "update_norm": 0.764352560043335, "read_norm": 0.5137579441070557 }, { "state_norm": 0.11622980982065201, "update_norm": 0.6543875932693481, "read_norm": 0.5391327142715454 }, { "state_norm": 0.11485152691602707, "update_norm": 0.6875792741775513, "read_norm": 0.49508407711982727 }, { "state_norm": 0.09085177630186081, "update_norm": 0.47813138365745544, "read_norm": 0.5154584646224976 } ] }, { "mean": 1.7542085275053978, "std": 1.03264163220941, "per_seed": [ 1.2069628238677979, 1.1996299028396606, 1.1370363235473633, 0.9978018403053284, 0.8372374773025513, 2.680882692337036, 1.9308146238327026, 4.043302536010742 ], "n": 8, "cfg": { "lr": 0.006, "beta": 0.5 }, "mechanism_signatures": [ { "state_norm": 0.10249067097902298, "update_norm": 0.49012985825538635, "read_norm": 0.5499802827835083 }, { "state_norm": 0.10208044946193695, "update_norm": 0.32191911339759827, "read_norm": 0.7767879366874695 }, { "state_norm": 0.12325321137905121, "update_norm": 0.5648367404937744, "read_norm": 0.6759800910949707 }, { "state_norm": 0.12312870472669601, "update_norm": 0.4876507818698883, "read_norm": 0.6405466794967651 }, { "state_norm": 0.10362966358661652, "update_norm": 0.5586110353469849, "read_norm": 0.560613751411438 }, { "state_norm": 0.11490625888109207, "update_norm": 0.3960989713668823, "read_norm": 0.5643085241317749 }, { "state_norm": 0.10144433379173279, "update_norm": 0.49015238881111145, "read_norm": 0.6192492246627808 }, { "state_norm": 0.07356467097997665, "update_norm": 0.2797723710536957, "read_norm": 0.5559394359588623 } ] } ], "parameter_counts": { "baseline": 71169, "idea": 79490 }, "confirmed": true }, "protocol_note": "Official bench sequence track; baseline and idea share transformer encoder/head and paired datasets." }