Weighted Resolvent-Equivariant Attention / bench_report.json

Unverified

Raw ⬇ ZIP
  1{
  2  "bench_version": 1,
  3  "track": "sequence",
  4  "model": "transformer_tiny",
  5  "metric_direction": "lower is better",
  6  "n_seeds": 8,
  7  "baseline": {
  8    "best_cfg": {
  9      "lr": 0.001,
 10      "epochs": 8,
 11      "lam": 0.0
 12    },
 13    "sweep": [
 14      {
 15        "cfg": {
 16          "lr": 0.001,
 17          "epochs": 8,
 18          "lam": 0.0
 19        },
 20        "mean": 0.19084082171320915
 21      },
 22      {
 23        "cfg": {
 24          "lr": 0.003,
 25          "epochs": 8,
 26          "lam": 0.0
 27        },
 28        "mean": 0.2167150229215622
 29      },
 30      {
 31        "cfg": {
 32          "lr": 0.006,
 33          "epochs": 8,
 34          "lam": 0.0
 35        },
 36        "mean": 0.26577289402484894
 37      }
 38    ],
 39    "full": {
 40      "mean": 0.19580462761223316,
 41      "std": 0.019276107610397688,
 42      "per_seed": [
 43        0.21272796392440796,
 44        0.16291894018650055,
 45        0.18857084214687347,
 46        0.19914554059505463,
 47        0.2006099969148636,
 48        0.1974794715642929,
 49        0.22919276356697083,
 50        0.17579150199890137
 51      ],
 52      "n": 8
 53    }
 54  },
 55  "idea": {
 56    "mean": 0.19599766097962856,
 57    "std": 0.018529453120078188,
 58    "per_seed": [
 59      0.21339234709739685,
 60      0.1631648987531662,
 61      0.19162587821483612,
 62      0.19664713740348816,
 63      0.1998937577009201,
 64      0.20089472830295563,
 65      0.22621849179267883,
 66      0.1761440485715866
 67    ],
 68    "n": 8
 69  },
 70  "comparison": {
 71    "delta_mean": 0.000193033367395401,
 72    "idea_wins": 3,
 73    "n_pairs": 8,
 74    "per_seed_diffs": [
 75      0.0006643831729888916,
 76      0.0002459585666656494,
 77      0.0030550360679626465,
 78      -0.0024984031915664673,
 79      -0.0007162392139434814,
 80      0.0034152567386627197,
 81      -0.002974271774291992,
 82      0.0003525465726852417
 83    ],
 84    "p_value": 0.7309,
 85    "mde": 0.0019204674874472793,
 86    "mde_rel_pct": 0.9808080181079926,
 87    "verdict": "no measurable effect",
 88    "system_worked": false
 89  },
 90  "mechanism_signature": {
 91    "prediction": "commutator regularization lowers trained attention commutator; multi-step compatibility follows from resolvent identity",
 92    "baseline_comm_rms": {
 93      "mean": 0.03365310910157859,
 94      "std": 0.005540028408003687,
 95      "per_seed": [
 96        0.028813593089580536,
 97        0.04024209827184677,
 98        0.037993812933564186,
 99        0.02756293211132288
100      ],
101      "n": 4
102    },
103    "idea_comm_rms": {
104      "mean": 0.02128551318310201,
105      "std": 0.0034537152021512133,
106      "per_seed": [
107        0.018550011795014143,
108        0.024316378869116306,
109        0.02507482608780265,
110        0.01720083598047495
111      ],
112      "n": 4
113    },
114    "baseline_function_reversal_mse": {
115      "mean": 1.6439854502677917,
116      "std": 0.2084291753940242,
117      "per_seed": [
118        1.8240505456924438,
119        1.5404354333877563,
120        1.3530447483062744,
121        1.8584110736846924
122      ],
123      "n": 4
124    },
125    "idea_function_reversal_mse": {
126      "mean": 1.6318478286266327,
127      "std": 0.20953659306825898,
128      "per_seed": [
129        1.841636061668396,
130        1.5250027179718018,
131        1.340692400932312,
132        1.820060133934021
133      ],
134      "n": 4
135    },
136    "observed_reduction_factor": 1.5810334856429433,
137    "confirmed": true,
138    "math_check": {
139      "resolvent_identity_rel_error": 4.1222850040650605e-15,
140      "gamma_1_commutator": 0.0,
141      "identity_confirmed": true
142    },
143    "idea_pilot_sweep": [
144      {
145        "cfg": {
146          "lr": 0.001,
147          "epochs": 8,
148          "lam": 0.3
149        },
150        "result": {
151          "mean": 0.19120756536722183,
152          "std": 0.018085350569029532,
153          "per_seed": [
154            0.21339234709739685,
155            0.1631648987531662,
156            0.19162587821483612,
157            0.19664713740348816
158          ],
159          "n": 4
160        }
161      },
162      {
163        "cfg": {
164          "lr": 0.003,
165          "epochs": 8,
166          "lam": 0.3
167        },
168        "result": {
169          "mean": 0.21730473637580872,
170          "std": 0.029589204134821655,
171          "per_seed": [
172            0.2179270088672638,
173            0.1916452795267105,
174            0.1942574828863144,
175            0.26538917422294617
176          ],
177          "n": 4
178        }
179      },
180      {
181        "cfg": {
182          "lr": 0.006,
183          "epochs": 8,
184          "lam": 0.3
185        },
186        "result": {
187          "mean": 0.26101431995630264,
188          "std": 0.014287884388682429,
189          "per_seed": [
190            0.2409307062625885,
191            0.26482611894607544,
192            0.2576110064983368,
193            0.28068944811820984
194          ],
195          "n": 4
196        }
197      }
198    ]
199  },
200  "idea_sweep": [
201    {
202      "cfg": {
203        "lr": 0.001,
204        "epochs": 8,
205        "lam": 0.3
206      },
207      "mean": 0.19120756536722183
208    },
209    {
210      "cfg": {
211        "lr": 0.003,
212        "epochs": 8,
213        "lam": 0.3
214      },
215      "mean": 0.21730473637580872
216    },
217    {
218      "cfg": {
219        "lr": 0.006,
220        "epochs": 8,
221        "lam": 0.3
222      },
223      "mean": 0.26101431995630264
224    }
225  ]
226}