Order-Adaptive Integral Optimizer / bench_report.json

Failed on benchmark

Raw ⬇ ZIP
  1{
  2  "bench_version": 1,
  3  "track": "tabular",
  4  "model": "mlp_tiny",
  5  "metric_direction": "lower is better",
  6  "n_seeds": 8,
  7  "baseline": {
  8    "best_cfg": {
  9      "lr": 0.003,
 10      "momentum": 0.9
 11    },
 12    "sweep": [
 13      {
 14        "cfg": {
 15          "lr": 0.003,
 16          "momentum": 0.0
 17        },
 18        "mean": 11.063072919845581
 19      },
 20      {
 21        "cfg": {
 22          "lr": 0.003,
 23          "momentum": 0.9
 24        },
 25        "mean": 8.369025349617004
 26      },
 27      {
 28        "cfg": {
 29          "lr": 0.01,
 30          "momentum": 0.0
 31        },
 32        "mean": 27.549113154411316
 33      },
 34      {
 35        "cfg": {
 36          "lr": 0.01,
 37          "momentum": 0.9
 38        },
 39        "mean": 21.363237619400024
 40      },
 41      {
 42        "cfg": {
 43          "lr": 0.03,
 44          "momentum": 0.0
 45        },
 46        "mean": 24.645026683807373
 47      },
 48      {
 49        "cfg": {
 50          "lr": 0.03,
 51          "momentum": 0.9
 52        },
 53        "mean": 24.67656421661377
 54      }
 55    ],
 56    "full": {
 57      "mean": 8.462080538272858,
 58      "std": 0.763596092727506,
 59      "per_seed": [
 60        9.01702880859375,
 61        8.495253562927246,
 62        8.408374786376953,
 63        7.555444240570068,
 64        7.235288619995117,
 65        8.890181541442871,
 66        9.808090209960938,
 67        8.286982536315918
 68      ],
 69      "n": 8
 70    }
 71  },
 72  "idea": {
 73    "mean": 11.567220091819763,
 74    "std": 1.9770478862735161,
 75    "per_seed": [
 76      11.564115524291992,
 77      11.624547958374023,
 78      11.304121017456055,
 79      9.648195266723633,
 80      8.412535667419434,
 81      11.701919555664062,
 82      12.67589282989502,
 83      15.606432914733887
 84    ],
 85    "n": 8
 86  },
 87  "comparison": {
 88    "delta_mean": 3.1051395535469055,
 89    "idea_wins": 0,
 90    "n_pairs": 8,
 91    "per_seed_diffs": [
 92      2.547086715698242,
 93      3.1292943954467773,
 94      2.8957462310791016,
 95      2.0927510261535645,
 96      1.1772470474243164,
 97      2.8117380142211914,
 98      2.867802619934082,
 99      7.319450378417969
100    ],
101    "p_value": 0.0081,
102    "mde": 1.515669973753461,
103    "mde_rel_pct": 17.91131586255045,
104    "verdict": "idea worse (significant)",
105    "system_worked": false
106  },
107  "protocol_note": "baseline sweep uses all lr values tried by idea and sweeps momentum; idea uses three lr settings.",
108  "mechanism_signature": {
109    "prediction": "on easy well-conditioned tabular training adaptive order remains mostly p=0 while residual contracts",
110    "observed_activation_fraction": 1.0,
111    "per_seed": [
112      {
113        "seed": 0,
114        "activated": true,
115        "first_activation": 5,
116        "grad_ratio": 0.42842475212727243
117      },
118      {
119        "seed": 1,
120        "activated": true,
121        "first_activation": 5,
122        "grad_ratio": 0.3283609454108231
123      },
124      {
125        "seed": 2,
126        "activated": true,
127        "first_activation": 5,
128        "grad_ratio": 0.3528653471373228
129      },
130      {
131        "seed": 3,
132        "activated": true,
133        "first_activation": 5,
134        "grad_ratio": 0.2514142861201013
135      },
136      {
137        "seed": 4,
138        "activated": true,
139        "first_activation": 5,
140        "grad_ratio": 0.27445735401524274
141      },
142      {
143        "seed": 5,
144        "activated": true,
145        "first_activation": 5,
146        "grad_ratio": 0.2794134775570858
147      },
148      {
149        "seed": 6,
150        "activated": true,
151        "first_activation": 5,
152        "grad_ratio": 0.36328053123347775
153      },
154      {
155        "seed": 7,
156        "activated": true,
157        "first_activation": 5,
158        "grad_ratio": 0.3765249020584892
159      }
160    ],
161    "confirmed": false
162  },
163  "idea_candidates": [
164    {
165      "cfg": {
166        "lr": 0.003,
167        "momentum": 0.9
168      },
169      "result": {
170        "mean": 11.567220091819763,
171        "std": 1.9770478862735161,
172        "per_seed": [
173          11.564115524291992,
174          11.624547958374023,
175          11.304121017456055,
176          9.648195266723633,
177          8.412535667419434,
178          11.701919555664062,
179          12.67589282989502,
180          15.606432914733887
181        ],
182        "n": 8
183      }
184    },
185    {
186      "cfg": {
187        "lr": 0.01,
188        "momentum": 0.9
189      },
190      "result": {
191        "mean": 18.361396193504333,
192        "std": 10.464071364766456,
193        "per_seed": [
194          41.432411193847656,
195          12.722975730895996,
196          24.072275161743164,
197          7.684035301208496,
198          11.690810203552246,
199          15.437067985534668,
200          24.278358459472656,
201          9.573235511779785
202        ],
203        "n": 8
204      }
205    },
206    {
207      "cfg": {
208        "lr": 0.03,
209        "momentum": 0.9
210      },
211      "result": {
212        "mean": 28.241891384124756,
213        "std": 10.055774240316543,
214        "per_seed": [
215          52.05145263671875,
216          23.101425170898438,
217          16.589689254760742,
218          23.7571964263916,
219          29.247011184692383,
220          33.055755615234375,
221          23.479705810546875,
222          24.652894973754883
223        ],
224        "n": 8
225      }
226    }
227  ]
228}