# Стенд-проверка (stage-2) · промт оператора: (универсальный) Ответ агента: { "worked": false, "confidence": 9, "verdict": "Ran the full 8-seed paired benchmark on the registered two_view_gauge_localization_v2 track using identical mlp_tiny systems and a tuned baseline sweep. The coarse-to-fine variant had higher test MSE (1.1728449 vs 1.1499638), paired delta +0.0228810, and permutation p=0.12625, so there is no significant win. The trained-model mechanism signature was also not confirmed: refinement slightly increased MAE from 0.9011721 to 0.9014216.", "metrics": { "baseline": "Best lr=0.01; full 8-seed MSE mean 1.1499638408 ± 0.1032812345", "idea": "lr=0.01; full 8-seed MSE mean 1.1728448644 ± 0.1088541057; paired delta +0.0228810236; p=0.12625" }, "how_to_run": "/home/maxwelhelp/main/bin/python3 registered_bench.py", "files": [ "registered_bench.py", "registered_bench_report.json", "registered_bench_stdout.json" ], "limitations": "The registered localization track predicts a scalar coordinate rather than a variable number of sources, so it only partially exercises the proposed neural slot-head use case. The intervention is a differentiable soft coarse-grid contraction, not the full learned-subspace projection-residual score with analytic Fourier gradient, fixed kappa^-2 refinement, NMS, and runtime evaluation accounting. Baseline sweep used three shared learning rates and the idea included the best plus two nearby settings; no larger architecture or additional localization tracks were tested.", "bench_report": { "bench_version": 1, "track": "two_view_gauge_localization_v2", "model": "mlp_tiny", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": { "lr": 0.01 }, "sweep": [ { "cfg": { "lr": 0.001 }, "mean": 1.2723695039749146 }, { "cfg": { "lr": 0.003 }, "mean": 1.2335557341575623 }, { "cfg": { "lr": 0.01 }, "mean": 1.0947982370853424 } ], "full": { "mean": 1.149963840842247, "std": 0.1032812344962866, "per_seed": [ 1.0112413167953491, 1.0117237567901611, 1.1340291500091553, 1.222198724746704, 1.1495081186294556, 1.266226053237915, 1.3084473609924316, 1.0963362455368042 ], "n": 8 } }, "idea": { "mean": 1.1728448644280434, "std": 0.10885410569769803, "per_seed": [ 0.9953804612159729, 1.0588462352752686, 1.2159733772277832, 1.2204293012619019, 1.1976563930511475, 1.2475311756134033, 1.355494499206543, 1.0914474725723267 ], "n": 8 }, "comparison": { "delta_mean": 0.022881023585796356, "idea_wins": 4, "n_pairs": 8, "per_seed_diffs": [ -0.01586085557937622, 0.04712247848510742, 0.08194422721862793, -0.001769423484802246, 0.048148274421691895, -0.01869487762451172, 0.04704713821411133, -0.004888772964477539 ], "p_value": 0.12625, "mde": 0.03145608783128045, "mde_rel_pct": 2.7353979937527115, "verdict": "no significant win", "system_worked": false }, "mechanism_signature": { "prediction": "coarse proposal plus fixed refinement should reduce raw localization error", "raw_mae_mean": 0.9011721238493919, "refined_mae_mean": 0.9014216214418411, "mean_proposal_shift": 0.017849167110398412, "confirmed": false }, "idea_nearby_sweep": { "0.001": { "mean": 1.2723439633846283, "std": 0.07400079392237602, "per_seed": [ 1.1923704147338867, 1.2113070487976074, 1.3731907606124878, 1.3125076293945312 ], "n": 4 }, "0.01": { "mean": 1.1226573437452316, "std": 0.098156125855815, "per_seed": [ 0.9953804612159729, 1.0588462352752686, 1.2159733772277832, 1.2204293012619019 ], "n": 4 } }, "custom_track": { "name": "two_view_gauge_localization_v2", "registered": true, "domain": "geometry/localization" } }, "system_verdict": "failed", "practical_verdict": "inconclusive", "mechanism_ok": 0, "system_judged": true }