{ "seed": 1267, "math_check": { "stationarity_gradient_norm": 6.938893903907227e-13, "hessian_error_norm": 1.5476311098377395e-07, "analytic_u": [ 0.14427515036590355, -0.24895276547974993 ] }, "toy_predictions": { "lambdas": [ 0.0, 0.1, 0.3, 1.0, 3.0, 10.0, 30.0, 100.0 ], "solutions": [ [ 2.0, 2.0 ], [ 1.4285714285714286, 1.9512195121951221 ], [ 0.9090909090909091, 1.8604651162790697 ], [ 0.4, 1.6 ], [ 0.15384615384615385, 1.1428571428571428 ], [ 0.04878048780487805, 0.5714285714285714 ], [ 0.01652892561983471, 0.23529411764705882 ], [ 0.004987531172069825, 0.07692307692307693 ] ], "trajectory_kl": [ 8.5, 4.557539851157596, 2.0855588680948824, 0.6400000000000001, 0.21060258422895783, 0.04557539851157595, 0.0074668259892055515, 0.0007893959047989377 ], "prediction_1_kl_monotone": true, "prediction_2_expected_lambda_u_tail": [ 0.5, 8.0 ], "prediction_2_observed_lambda_u_tail": [ 0.4941419212836014, 6.821805645335057 ], "prediction_2_relative_error": 0.1472742943330995, "prediction_3_predicted_curvature_ratio": 4.0, "prediction_3_observed_curvature_ratio": 4.0, "prediction_3_relative_error": 0.0 }, "policy_experiment": { "device": "cuda", "lambda": 2.0, "action_regularization": { "task_mse_half": 0.3949859142303467, "action_penalty": 0.049373235553503036, "trajectory_kl": 0.1613738238811493, "W": [ [ 0.20000000298023224, -0.07999999821186066 ], [ 0.03999998793005943, 0.1599999964237213 ], [ -0.13999998569488525, 0.06000000238418579 ] ] }, "noise_whitened_drift_regularization": { "task_mse_half": 0.411968857049942, "action_penalty": 0.11426855623722076, "trajectory_kl": 0.037015583366155624, "W": [ [ 0.0476190522313118, -0.20000000298023224 ], [ 0.00952380895614624, 0.4000000059604645 ], [ -0.03333333507180214, 0.15000000596046448 ] ] } } }