{"id":"W2800222226","doi":"10.7939/r3cw3c","title":"A general framework for reducing variance in agent evaluation","year":2010,"lang":"en","type":"article","venue":"University of Alberta Library","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Variance (accounting); Computer science; Risk analysis (engineering); Business","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001489238,0.00006207285,0.00009060105,0.0001027242,0.00006331188,0.00003278918,0.000614404,0.00007962764,0.0001456097],"category_scores_gemma":[0.0001131898,0.00007680077,0.0000436357,0.0002176861,0.00003247312,0.0009627063,0.0001865078,0.0001599268,0.000009063375],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001526388,"about_ca_system_score_gemma":0.0001342612,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001981034,"about_ca_topic_score_gemma":0.00002587429,"domain_scores_codex":[0.9993781,0.0000377479,0.00009683194,0.000198663,0.0001512334,0.0001374191],"domain_scores_gemma":[0.9992855,0.000210984,0.00009795797,0.0003361754,0.00002697019,0.00004240828],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00004077416,0.00006487267,0.007965283,0.00004059968,0.00002736037,0.00000516051,0.01037146,0.3132281,0.0008097534,0.6571007,0.002854405,0.007491557],"study_design_scores_gemma":[0.0004147875,0.0000511054,0.006933718,0.00004311135,0.00000783114,0.000001077314,0.00005564419,0.977514,0.0004739488,0.006521263,0.00787233,0.0001112204],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1412317,0.000007410466,0.8504765,0.001466693,0.0004338412,0.0002414131,3.054834e-7,0.00002839194,0.006113741],"genre_scores_gemma":[0.3651924,0.000005620115,0.6310846,0.00009973229,0.00004343355,4.27785e-7,0.000009661448,0.000005704277,0.00355847],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.6642858,"threshold_uncertainty_score":0.3131845,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01671819584397845,"score_gpt":0.2321567703057499,"score_spread":0.2154385744617714,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}