Coupled Workload-Order Gate / report_bench_2026-08-31T193928.md

Failed on benchmark

Raw ⬇ ZIP

Стенд-проверка (stage-2) · промт оператора:

(универсальный)

Ответ агента:

{ "worked": false, "confidence": 9, "verdict": "Implemented and evaluated the Coupled Workload-Order Gate on the structurally matched registered dynamics track with identical rnn_small systems, tuned baseline learning rate, shared hyperparameter union, and 8 paired seeds. The best idea configuration was statistically indistinguishable from the tuned baseline (delta_mean=-6.273621693253517e-7, permutation p=0.9473), and the corrected trained-model mechanism signature was not confirmed.", "metrics": { "baseline": "Best lr=0.006; 8-seed mean test MSE=0.002105819425196387.", "idea": "Best lr=0.006, lambda=0.2; 8-seed mean test MSE=0.0021051920630270615; paired delta=-6.273621693253517e-7; 3/8 wins; permutation p=0.9473; no measurable effect." }, "bench_report": { "bench_version": 1, "track": "dynamics", "model": "rnn_small", "metric_direction": "lower is better", "n_seeds": 8, "baseline": { "best_cfg": [ 0.006 ], "sweep": [ { "config": [ 0.001 ], "per_seed": [ 0.007754037156701088, 0.003745198715478182, 0.0025475025177001953, 0.003148377873003483, 0.004954223055392504, 0.004425588063895702, 0.0044244141317903996, 0.009162315167486668 ], "mean": 0.005020207085181028, "std": 0.002286214586911066 }, { "config": [ 0.003 ], "per_seed": [ 0.002470282604917884, 0.002755755092948675, 0.0011578478151932359, 0.007462490815669298, 0.0025412666145712137, 0.0030948929488658905, 0.0021244161762297153, 0.0019843443296849728 ], "mean": 0.0029489120497601107, "std": 0.0019140223046789351 }, { "config": [ 0.006 ], "per_seed": [ 0.002004395006224513, 0.001437882543541491, 0.0015748648438602686, 0.002295783953741193, 0.0024020641576498747, 0.0024898042902350426, 0.0020851334556937218, 0.0025566271506249905 ], "mean": 0.002105819425196387, "std": 0.0004163252013427006 } ], "full": { "config": [ 0.006 ], "per_seed": [ 0.002004395006224513, 0.001437882543541491, 0.0015748648438602686, 0.002295783953741193, 0.0024020641576498747, 0.0024898042902350426, 0.0020851334556937218, 0.0025566271506249905 ], "mean": 0.002105819425196387, "std": 0.0004163252013427006 } }, "idea": { "best_cfg": [ 0.006, 0.2 ], "sweep": [ { "config": [ 0.001, 0.02 ], "per_seed": [ 0.007752448320388794, 0.003742651315405965, 0.0025358593557029963, 0.003136643208563328, 0.004953413270413876, 0.004411333240568638, 0.0043915919959545135, 0.00907886028289795 ], "mean": 0.005000350123737007, "std": 0.0022695797131772446, "lambda": 0.02 }, { "config": [ 0.003, 0.02 ], "per_seed": [ 0.0024700742214918137, 0.002753326902166009, 0.0011541355634108186, 0.007465476635843515, 0.0025420342572033405, 0.003092885483056307, 0.002128961030393839, 0.0019743957091122866 ], "mean": 0.002947661225334741, "std": 0.0019159624457765199, "lambda": 0.02 }, { "config": [ 0.006, 0.02 ], "per_seed": [ 0.00200583110563457, 0.0014351154677569866, 0.0015823777066543698, 0.002294273115694523, 0.0024029696360230446, 0.002487938152626157, 0.002087509259581566, 0.002553937491029501 ], "mean": 0.0021062439918750897, "std": 0.000414867290214433, "lambda": 0.02 }, { "config": [ 0.001, 0.08 ], "per_seed": [ 0.007747639436274767, 0.003735017729923129, 0.0025040435139089823, 0.0031060727778822184, 0.004950913600623608, 0.004369518253952265, 0.004298955202102661, 0.008836028166115284 ], "mean": 0.004943523585097864, "std": 0.0022215643665023208, "lambda": 0.08 }, { "config": [ 0.003, 0.08 ], "per_seed": [ 0.002469458617269993, 0.0027461235877126455, 0.0011356071336194873, 0.007463372778147459, 0.002544374903663993, 0.003089339705184102, 0.0021439625415951014, 0.0019459263421595097 ], "mean": 0.0029422707011690363, "std": 0.0019189420653264089, "lambda": 0.08 }, { "config": [ 0.006, 0.08 ], "per_seed": [ 0.002010113326832652, 0.0014257561415433884, 0.0015752587753790617, 0.0023342787753790617, 0.0024051156360656023, 0.002479771850630641, 0.0020939703099429607, 0.0025461555924266577 ], "mean": 0.002108802509610541, "std": 0.0004189643943471674, "lambda": 0.08 }, { "config": [ 0.001, 0.2 ], "per_seed": [ 0.007737650536000729, 0.003719724016264081, 0.0024441867135465145, 0.0030418795067816973, 0.004945157095789909, 0.004281151108443737, 0.004141305573284626, 0.008383535780012608 ], "mean": 0.004836823791265488, "std": 0.0021378448640202853, "lambda": 0.2 }, { "config": [ 0.003, 0.2 ], "per_seed": [ 0.0024682374205440283, 0.0027322175446897745, 0.001090307254344225, 0.0074797505512833595, 0.002549185650423169, 0.0030852353665977716, 0.0021790594328194857, 0.0018948764773085713 ], "mean": 0.002934858712251298, "std": 0.001932480390978357, "lambda": 0.2 }, { "config": [ 0.006, 0.2 ], "per_seed": [ 0.0020184817258268595, 0.0014025752898305655, 0.0015944599872455, 0.0023251562379300594, 0.0024057631380856037, 0.0024610732216387987, 0.0021019575651735067, 0.0025320693384855986 ], "mean": 0.0021051920630270615, "std": 0.00041570070698645627, "lambda": 0.2 } ], "per_seed": [ 0.0020184817258268595, 0.0014025752898305655, 0.0015944599872455, 0.0023251562379300594, 0.0024057631380856037, 0.0024610732216387987, 0.0021019575651735067, 0.0025320693384855986 ], "mean": 0.0021051920630270615, "std": 0.00041570070698645627 }, "comparison": { "delta_mean": -6.273621693253517e-07, "idea_wins": 3, "n_pairs": 8, "per_seed_diffs": [ 1.408671960234642e-05, -3.530725371092558e-05, 1.9595143385231495e-05, 2.9372284188866615e-05, 3.698980435729027e-06, -2.873106859624386e-05, 1.6824109479784966e-05, -2.45578121393919e-05 ], "p_value": 0.9473, "mde": 2.0996531702215405e-05, "mde_rel_pct": 0.9970718025956707, "verdict": "no measurable effect", "system_worked": false }, "mechanism_signature": { "track_match": "stability/control -> dynamics", "prediction": "order penalty should reduce positive controlled-minus-baseline workload events", "signature": { "predicted_mean_violation": 2.5829562929402528e-05, "observed_mean_violation": 1.9965890448503387e-05, "predicted_event_rate": 0.004999999888241291, "observed_event_rate": 0.004687499924330041, "confirmed": false } } }, "how_to_run": "/home/maxwelhelp/main/bin/python3 bench_order_gate.py", "files": [ "bench_order_gate.py", "bench_report.json" ], "limitations": "Only the registered built-in dynamics pendulum task was tested; no production admission or MoE workload traces, serving latency, throughput-at-equal-admission, p99 latency, or dropped-token metric was evaluated. The workload surrogate used model predictions and targets rather than measured service costs. The corrected mechanism signature was not confirmed.", "system_verdict": "failed", "practical_verdict": "no_effect", "mechanism_ok": 0, "system_judged": true }