Sensitivity-Conditioned Neural ODE Pruning / bench_report.json
Failed on benchmark
1{
2 "bench_version": 1,
3 "track": "dynamics",
4 "model": "rnn_small",
5 "metric_direction": "lower is better",
6 "n_seeds": 8,
7 "baseline": {
8 "best_cfg": {
9 "lr": 0.006,
10 "epochs": 15,
11 "keep": 32,
12 "method": "magnitude"
13 },
14 "sweep": [
15 {
16 "cfg": {
17 "lr": 0.0015,
18 "epochs": 15,
19 "keep": 32,
20 "method": "magnitude"
21 },
22 "mean": 0.0010415702417958528
23 },
24 {
25 "cfg": {
26 "lr": 0.003,
27 "epochs": 15,
28 "keep": 32,
29 "method": "magnitude"
30 },
31 "mean": 0.0007543887477368116
32 },
33 {
34 "cfg": {
35 "lr": 0.006,
36 "epochs": 15,
37 "keep": 32,
38 "method": "magnitude"
39 },
40 "mean": 0.0004434098445926793
41 }
42 ],
43 "full": {
44 "mean": 0.0004183898890914861,
45 "std": 0.00018028736524803996,
46 "per_seed": [
47 0.0002669677196536213,
48 0.00046067332732491195,
49 0.0002578582789283246,
50 0.0007881400524638593,
51 0.0002825864066835493,
52 0.0005642541800625622,
53 0.0002470404433552176,
54 0.00047959870425984263
55 ],
56 "n": 8
57 }
58 },
59 "idea": {
60 "best_cfg": {
61 "lr": 0.006,
62 "epochs": 15,
63 "keep": 32,
64 "method": "sensitivity"
65 },
66 "sweep": [
67 {
68 "cfg": {
69 "lr": 0.0015,
70 "epochs": 15,
71 "keep": 32,
72 "method": "sensitivity"
73 },
74 "mean": 0.0051941196143161505
75 },
76 {
77 "cfg": {
78 "lr": 0.003,
79 "epochs": 15,
80 "keep": 32,
81 "method": "sensitivity"
82 },
83 "mean": 0.0029561036499217153
84 },
85 {
86 "cfg": {
87 "lr": 0.006,
88 "epochs": 15,
89 "keep": 32,
90 "method": "sensitivity"
91 },
92 "mean": 0.0012721883977064863
93 }
94 ],
95 "mean": 0.0012721883977064863,
96 "std": 0.0006724726946615953,
97 "per_seed": [
98 0.0011970085324719548,
99 0.0006662395317107439,
100 0.0011209469521418214,
101 0.0028435320127755404,
102 0.001395156024955213,
103 0.0005800570943392813,
104 0.0015368196181952953,
105 0.0008377474150620401
106 ],
107 "n": 8
108 },
109 "comparison": {
110 "delta_mean": 0.0008537985086150002,
111 "idea_wins": 0,
112 "n_pairs": 8,
113 "per_seed_diffs": [
114 0.0009300408128183335,
115 0.00020556620438583195,
116 0.0008630886732134968,
117 0.002055391960311681,
118 0.0011125696182716638,
119 1.5802914276719093e-05,
120 0.0012897791748400778,
121 0.00035814871080219746
122 ],
123 "p_value": 0.0081,
124 "mde": 0.000554180300336378,
125 "mde_rel_pct": 132.45547150762997,
126 "verdict": "idea worse (significant)",
127 "system_worked": false
128 },
129 "mechanism_signature": {
130 "prediction": "weighted sensitivity information and incremental residual rank identify useful hidden groups",
131 "per_seed": [
132 {
133 "retained_units": 32,
134 "params_total": 13313,
135 "mean_information_retained": 9.77945613861084,
136 "mean_residual_retained": 3.6348969729260716e-05,
137 "importance_ablation_corr": 0.477768394227869,
138 "predicted_information": 11.743715286254883,
139 "observed_ablation": 0.000710174660306643
140 },
141 {
142 "retained_units": 32,
143 "params_total": 13313,
144 "mean_information_retained": 9.776529312133789,
145 "mean_residual_retained": 5.1056572374363896e-05,
146 "importance_ablation_corr": 0.42190104414981555,
147 "predicted_information": 11.318390846252441,
148 "observed_ablation": 0.0006673403675832024
149 },
150 {
151 "retained_units": 32,
152 "params_total": 13313,
153 "mean_information_retained": 9.697711944580078,
154 "mean_residual_retained": 2.9330890470191662e-05,
155 "importance_ablation_corr": 0.47564845830287067,
156 "predicted_information": 10.515939712524414,
157 "observed_ablation": 0.0006880304698516637
158 },
159 {
160 "retained_units": 32,
161 "params_total": 13313,
162 "mean_information_retained": 8.47030258178711,
163 "mean_residual_retained": 4.5788009657599105e-05,
164 "importance_ablation_corr": 0.5436690158311026,
165 "predicted_information": 10.03992748260498,
166 "observed_ablation": 0.000801966964582998
167 },
168 {
169 "retained_units": 32,
170 "params_total": 13313,
171 "mean_information_retained": 6.386395454406738,
172 "mean_residual_retained": 1.4392690587783363e-05,
173 "importance_ablation_corr": 0.18481644015160745,
174 "predicted_information": 9.996113777160645,
175 "observed_ablation": 0.0005640771121928623
176 },
177 {
178 "retained_units": 32,
179 "params_total": 13313,
180 "mean_information_retained": 10.290732383728027,
181 "mean_residual_retained": 3.206331243177374e-05,
182 "importance_ablation_corr": 0.5280788239090402,
183 "predicted_information": 11.632051467895508,
184 "observed_ablation": 0.000697915483999556
185 },
186 {
187 "retained_units": 32,
188 "params_total": 13313,
189 "mean_information_retained": 11.150772094726562,
190 "mean_residual_retained": 5.351222108629372e-05,
191 "importance_ablation_corr": 0.5366510030845261,
192 "predicted_information": 12.404411315917969,
193 "observed_ablation": 0.0007341111947982149
194 },
195 {
196 "retained_units": 32,
197 "params_total": 13313,
198 "mean_information_retained": 9.854597091674805,
199 "mean_residual_retained": 3.187904701462685e-05,
200 "importance_ablation_corr": 0.38449025322316216,
201 "predicted_information": 12.160406112670898,
202 "observed_ablation": 0.000575491002093198
203 }
204 ],
205 "confirmed": false
206 },
207 "protocol_notes": {
208 "paired_seeds": [
209 0,
210 1,
211 2,
212 3,
213 4,
214 5,
215 6,
216 7
217 ],
218 "n_train": 400,
219 "n_test": 200,
220 "retained_units": 32,
221 "effective_unit_fraction": 0.5,
222 "baseline_method": "outgoing-head magnitude selection",
223 "idea_method": "trained-output sensitivity trace plus leave-one-group residual selection",
224 "custom_track": null
225 }
226}