Probe-Then-Partitioned Multi-Task Trunk / results.json
Failed on benchmark
1{
2 "seed": 1234,
3 "math_exact": {
4 "same": 1.000088900582341e-12,
5 "orthogonal": 1.0,
6 "antipodal": 1.999999999999
7 },
8 "noise_scaling": [
9 {
10 "sigma": 0.01,
11 "observed": 0.0006986228358126927,
12 "predicted": 0.0007000000000000001,
13 "ratio": 0.9980326225895608
14 },
15 {
16 "sigma": 0.03,
17 "observed": 0.006276764365943627,
18 "predicted": 0.006299999999999999,
19 "ratio": 0.9963118041180361
20 },
21 {
22 "sigma": 0.06,
23 "observed": 0.02462667338052535,
24 "predicted": 0.025199999999999997,
25 "ratio": 0.977248943671641
26 },
27 {
28 "sigma": 0.1,
29 "observed": 0.06612041432046092,
30 "predicted": 0.07,
31 "ratio": 0.944577347435156
32 },
33 {
34 "sigma": 0.18,
35 "observed": 0.19009649754592634,
36 "predicted": 0.2268,
37 "ratio": 0.8381679785975588
38 },
39 {
40 "sigma": 0.3,
41 "observed": 0.4062411889336691,
42 "predicted": 0.63,
43 "ratio": 0.644827284021697
44 }
45 ],
46 "density_sweep": [
47 {
48 "sigma": 0.01,
49 "within_distance": 0.0007520450488480197,
50 "between_distance": 1.0050393620475382,
51 "between_minus_within": 1.0042873169986901,
52 "cluster_purity": 1.0,
53 "mean_core": 0.0004516606447072108
54 },
55 {
56 "sigma": 0.05,
57 "within_distance": 0.02045960371044641,
58 "between_distance": 1.011824199526698,
59 "between_minus_within": 0.9913645958162517,
60 "cluster_purity": 1.0,
61 "mean_core": 0.013977510386516345
62 },
63 {
64 "sigma": 0.1,
65 "within_distance": 0.06391457595483079,
66 "between_distance": 0.8807575499937699,
67 "between_minus_within": 0.8168429740389391,
68 "cluster_purity": 1.0,
69 "mean_core": 0.0410638974190513
70 },
71 {
72 "sigma": 0.2,
73 "within_distance": 0.21223261844670432,
74 "between_distance": 1.0524008180311804,
75 "between_minus_within": 0.8401681995844761,
76 "cluster_purity": 1.0,
77 "mean_core": 0.1612017055260236
78 },
79 {
80 "sigma": 0.35,
81 "within_distance": 0.4622438689769046,
82 "between_distance": 0.8385216512010965,
83 "between_minus_within": 0.37627778222419184,
84 "cluster_purity": 1.0,
85 "mean_core": 0.3086342844447999
86 },
87 {
88 "sigma": 0.6,
89 "within_distance": 0.7013901853588502,
90 "between_distance": 0.8659499411151649,
91 "between_minus_within": 0.16455975575631465,
92 "cluster_purity": 1.0,
93 "mean_core": 0.4020908234366968
94 }
95 ],
96 "training": {
97 "angles": [
98 0.0,
99 0.1,
100 0.18,
101 0.25,
102 1.3,
103 1.4,
104 1.48,
105 1.58
106 ],
107 "discovered_groups": [
108 0,
109 0,
110 0,
111 0,
112 1,
113 1,
114 1,
115 1
116 ],
117 "probe_core": [
118 0.01615630721187855,
119 0.0049958347219741794,
120 0.00319829369738045,
121 0.011228922063957647,
122 0.016156307211878662,
123 0.0049958347219741794,
124 0.0049958347219742905,
125 0.016156307211878662
126 ],
127 "shared_population_loss": 0.3723075967863627,
128 "discovered_population_loss": 0.00958122769370327,
129 "oracle_population_loss": 0.00958122769370327,
130 "conflict_sweep": [
131 {
132 "angle": 0,
133 "shared_loss": 0.0,
134 "partition_loss": 0.0,
135 "observed_gain": 0.0,
136 "predicted_gain": 0.0
137 },
138 {
139 "angle": 0.2,
140 "shared_loss": 0.009966711079379185,
141 "partition_loss": 0.0,
142 "observed_gain": 0.009966711079379185,
143 "predicted_gain": 0.009966711079379187
144 },
145 {
146 "angle": 0.4,
147 "shared_loss": 0.03946950299855746,
148 "partition_loss": 0.0,
149 "observed_gain": 0.03946950299855746,
150 "predicted_gain": 0.03946950299855745
151 },
152 {
153 "angle": 0.6,
154 "shared_loss": 0.08733219254516086,
155 "partition_loss": 0.0,
156 "observed_gain": 0.08733219254516086,
157 "predicted_gain": 0.08733219254516084
158 },
159 {
160 "angle": 0.8,
161 "shared_loss": 0.15164664532641736,
162 "partition_loss": 1.3877787807814457e-17,
163 "observed_gain": 0.15164664532641736,
164 "predicted_gain": 0.1516466453264173
165 },
166 {
167 "angle": 1.0,
168 "shared_loss": 0.22984884706593012,
169 "partition_loss": 0.0,
170 "observed_gain": 0.22984884706593012,
171 "predicted_gain": 0.22984884706593012
172 },
173 {
174 "angle": 1.2,
175 "shared_loss": 0.31882112276166324,
176 "partition_loss": 0.0,
177 "observed_gain": 0.31882112276166324,
178 "predicted_gain": 0.3188211227616632
179 },
180 {
181 "angle": 1.5707963267948966,
182 "shared_loss": 0.5,
183 "partition_loss": 0.0,
184 "observed_gain": 0.5,
185 "predicted_gain": 0.49999999999999994
186 }
187 ]
188 },
189 "predictions": {
190 "P1": "cosine distance is 0/1/2 for same/orthogonal/antipodal normalized embeddings",
191 "P2": "for d=8 and small sigma, E[distance] ~= 7 sigma^2",
192 "P3": "for two unit linear tasks at angle theta, partition gain per task = (1-|cos(theta)|)/2"
193 }
194}