{"id":"W3156545714","doi":"10.1177/10597123221095880","title":"What’s a good prediction? Challenges in evaluating an agent’s knowledge","year":2022,"lang":"en","type":"article","venue":"Adaptive Behavior","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates; University of Alberta; Canada Research Chairs; DeepMind; Alberta Machine Intelligence Institute; Canadian Institute for Advanced Research","keywords":"Computer science; Artificial intelligence; Knowledge management; Cognitive science; Machine learning; Psychology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02507325,0.0008423746,0.001655171,0.001199211,0.001112313,0.006689593,0.001906432,0.003164715,0.001922237],"category_scores_gemma":[0.1164608,0.0003365976,0.0005097783,0.001080799,0.003589979,0.0105794,0.001970019,0.003460119,0.0006112986],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001624763,"about_ca_system_score_gemma":0.001929067,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003420724,"about_ca_topic_score_gemma":0.003577757,"domain_scores_codex":[0.981385,0.01093605,0.0009540912,0.001850733,0.004406636,0.0004675023],"domain_scores_gemma":[0.906543,0.07098107,0.00485658,0.007228232,0.008035792,0.00235534],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001276425,0.001027592,0.06656356,0.001286991,0.0006586082,0.0005770877,0.003147574,0.1714058,0.006449562,0.1371162,0.01532832,0.5951622],"study_design_scores_gemma":[0.0001616977,0.0009414081,0.01788406,0.0008884718,0.0001750434,0.0003400721,0.002844355,0.5539283,0.008893692,0.3987759,0.01494523,0.0002217429],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4040055,0.006368631,0.4906971,0.04694023,0.0005768188,0.0004374663,0.0007755281,0.001738384,0.04846025],"genre_scores_gemma":[0.9137957,0.0006494335,0.0837396,0.0007269628,0.00008723074,0.00008514636,0.0001667441,0.0001211583,0.0006279523],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02507325,"threshold_uncertainty_score":0.1326016,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1911370875658001,"score_gpt":0.3612687422696316,"score_spread":0.1701316547038315,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}