Persistent Workspace for Online Adaptation / report_bench_2026-09-01T145459.md

Mechanism confirmed, baseline not beaten

Raw ⬇ ZIP

Стенд-проверка (stage-2) · промт оператора:

(универсальный)

Ответ агента:

{ "worked": false, "confidence": 9, "verdict": "The persistent workspace was evaluated on the registered sequence forecasting track with 8 paired seeds and a tuned baseline sweep. The idea's best MSE was 0.165073 versus baseline 0.144882; paired delta was +0.020191 with permutation p=0.0081, so the idea was significantly worse. The trained-model mechanism signature was confirmed, but the task metric did not improve.", "metrics": { "baseline": "mean test MSE 0.1448821258, std 0.0113162536, best lr=0.006 retention=1.0", "idea": "mean test MSE 0.1650733370, std 0.0140552122, best lr=0.006 gate_bias=1.0" }, "how_to_run": "/home/maxwelhelp/main/bin/python3 stage2_bench.py", "files": [ "stage2_bench.py", "bench_report.json" ], "limitations": "Only the registered sequence track was tested; dynamics, longer horizons, larger datasets, GRU/SSM baselines, and distribution-switch adaptation were not tested.", "bench_report": { "bench_version": 1, "track": "sequence", "model": "persistent_workspace_shared_transition", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": { "lr": 0.006, "retention": 1.0 }, "sweep": [ { "cfg": { "lr": 0.001, "retention": 0.5 }, "mean": 0.4334259256720543 }, { "cfg": { "lr": 0.001, "retention": 0.8 }, "mean": 0.35411225259304047 }, { "cfg": { "lr": 0.001, "retention": 1.0 }, "mean": 0.3027268350124359 }, { "cfg": { "lr": 0.003, "retention": 0.5 }, "mean": 0.22246700897812843 }, { "cfg": { "lr": 0.003, "retention": 0.8 }, "mean": 0.16858474910259247 }, { "cfg": { "lr": 0.003, "retention": 1.0 }, "mean": 0.15582704544067383 }, { "cfg": { "lr": 0.006, "retention": 0.5 }, "mean": 0.1905057244002819 }, { "cfg": { "lr": 0.006, "retention": 0.8 }, "mean": 0.15795930102467537 }, { "cfg": { "lr": 0.006, "retention": 1.0 }, "mean": 0.14923853054642677 } ], "full": { "mean": 0.14488212577998638, "std": 0.01131625360567458, "per_seed": [ 0.1530856192111969, 0.13499000668525696, 0.15433596074581146, 0.15454253554344177, 0.1579333245754242, 0.12868644297122955, 0.14616775512695312, 0.1293153613805771 ], "n": 8 } }, "idea": { "best_cfg": { "lr": 0.006, "gate_bias": 1.0 }, "sweep": [ { "cfg": { "lr": 0.001, "gate_bias": -1.0 }, "mean": 0.46400970965623856 }, { "cfg": { "lr": 0.001, "gate_bias": 0.0 }, "mean": 0.4027043655514717 }, { "cfg": { "lr": 0.001, "gate_bias": 1.0 }, "mean": 0.35902588069438934 }, { "cfg": { "lr": 0.003, "gate_bias": -1.0 }, "mean": 0.2567196935415268 }, { "cfg": { "lr": 0.003, "gate_bias": 0.0 }, "mean": 0.22174151614308357 }, { "cfg": { "lr": 0.003, "gate_bias": 1.0 }, "mean": 0.17892902344465256 }, { "cfg": { "lr": 0.006, "gate_bias": -1.0 }, "mean": 0.20071889832615852 }, { "cfg": { "lr": 0.006, "gate_bias": 0.0 }, "mean": 0.1830645389854908 }, { "cfg": { "lr": 0.006, "gate_bias": 1.0 }, "mean": 0.16491885110735893 } ], "per_seed": [ 0.17458929121494293, 0.14810295403003693, 0.15635818243026733, 0.18062497675418854, 0.17810462415218353, 0.1477891057729721, 0.18171468377113342, 0.1533028781414032 ], "mean": 0.165073337033391, "std": 0.014055212196181685, "n": 8 }, "comparison": { "delta_mean": 0.020191211253404617, "idea_wins": 0, "n_pairs": 8, "per_seed_diffs": [ 0.021503672003746033, 0.013112947344779968, 0.0020222216844558716, 0.026082441210746765, 0.02017129957675934, 0.019102662801742554, 0.0355469286441803, 0.02398751676082611 ], "p_value": 0.0081, "mde": 0.008168416398466487, "mde_rel_pct": 5.6379738732373355, "verdict": "idea worse (significant)", "system_worked": false }, "mechanism_signature": { "quantity": "trained-state retention multiplier per macrostep", "predicted_mean_gate": 0.7511228322982788, "observed_mean_norm_ratio": 0.8821052908973657, "baseline_norm_ratios": [ 1.0832295149921198, 1.0580671611678902, 0.9077942589346728, 0.7774226224379195, 0.869266350595166 ], "idea_norms": [ 1.4678477048873901, 1.3970891237258911, 1.4260311126708984, 1.2615630626678467, 0.9309729337692261, 0.759112536907196 ], "confirmed": true } }, "system_verdict": "partial", "practical_verdict": "harms", "mechanism_ok": 1, "system_judged": true }