{"id":"W4399552717","doi":"10.1613/jair.1.15155","title":"Mitigating Value Hallucination in Dyna-Style Planning via Multistep Predecessor Models","year":2024,"lang":"en","type":"article","venue":"Journal of Artificial Intelligence Research","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Hallucinating; Computer science; Reinforcement learning; Bootstrapping (finance); Artificial intelligence; Value (mathematics); Function (biology); Successor cardinal; Sample (material); Bellman equation; Machine learning; Algorithm; Mathematical optimization; Econometrics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003201445,0.0008164245,0.0008886348,0.0004491805,0.0004760678,0.001219052,0.00177201,0.0009733555,0.001443506],"category_scores_gemma":[0.01453628,0.0006988541,0.0004763179,0.0003155374,0.001533966,0.002349951,0.002111769,0.002079114,0.0001829869],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008922886,"about_ca_system_score_gemma":0.001393517,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003296653,"about_ca_topic_score_gemma":0.003384313,"domain_scores_codex":[0.998614,0.0007230237,0.00006754394,0.0002018087,0.000279185,0.0001144951],"domain_scores_gemma":[0.9897579,0.007634704,0.0008915383,0.0009061063,0.0005019269,0.0003078031],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003300572,0.00009556548,0.002306578,0.0000820345,0.00007522167,0.000158998,0.000376556,0.9217743,0.002623946,0.02327395,0.0005300972,0.04837282],"study_design_scores_gemma":[0.0000165584,0.00005031199,0.00007159872,0.000004725849,0.000006570196,0.00001599892,0.00001247235,0.9910824,0.001053825,0.007463082,0.0002152041,0.000007153869],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1239066,0.0002182287,0.8723093,0.0003976877,0.00005079921,0.00006188005,0.00003325548,0.0008486332,0.002173522],"genre_scores_gemma":[0.9196666,0.00007904239,0.07873042,0.0001087574,0.00001713737,0.0000632394,0.00003194735,0.00005574018,0.001247096],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003296653,"threshold_uncertainty_score":0.01693106,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1987057900527737,"score_gpt":0.4453474910418657,"score_spread":0.246641700989092,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}