{"id":"W2800222226","doi":"10.7939/r3cw3c","title":"A general framework for reducing variance in agent evaluation","year":2010,"lang":"en","type":"article","venue":"University of Alberta Library","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Variance (accounting); Computer science; Risk analysis (engineering); Business","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01943447,0.002256971,0.00245933,0.002742369,0.001151873,0.003970855,0.003744526,0.002319187,0.004002336],"category_scores_gemma":[0.04765212,0.001060031,0.002389549,0.001727918,0.003082165,0.004846109,0.004776303,0.004614942,0.000982901],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002720442,"about_ca_system_score_gemma":0.002939193,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002487893,"about_ca_topic_score_gemma":0.002130683,"domain_scores_codex":[0.981299,0.009015297,0.0009548577,0.002385935,0.005587408,0.0007574596],"domain_scores_gemma":[0.9791152,0.01350799,0.001516092,0.002580822,0.002894361,0.0003854349],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001176411,0.0001413527,0.001659831,0.0002834566,0.0002629457,0.0001426136,0.0003685796,0.301862,0.003093793,0.5263813,0.003241917,0.1624446],"study_design_scores_gemma":[0.00004248449,0.0001492023,0.0005341549,0.00008306669,0.00006972213,0.00009656105,0.00003916409,0.736428,0.00212792,0.2540662,0.006309268,0.00005435073],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0007399538,0.0001654436,0.9976516,0.0001374353,0.00002152385,0.00005003281,0.00001481731,0.00008187601,0.001137222],"genre_scores_gemma":[0.1666729,0.0005900636,0.8273221,0.000432821,0.000306918,0.0007329733,0.0001415341,0.0002965735,0.003504124],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01943447,"threshold_uncertainty_score":0.1027805,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01671819584397845,"score_gpt":0.2321567703057499,"score_spread":0.2154385744617714,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}