{"id":"W3206265976","doi":"10.48550/arxiv.2107.00848","title":"Systematic Evaluation of Causal Discovery in Visual Model Based\\n Reinforcement Learning","year":2021,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Polytechnique Montréal","funders":"","keywords":"Causality (physics); Causal model; Causal structure; Reinforcement learning; Premise; Computer science; Modularity (biology); Artificial intelligence; Machine learning; Causal reasoning; Representation (politics); Psychology; Mathematics; Cognition","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01199066,0.001246468,0.001129222,0.001141265,0.0005911805,0.001371734,0.002824659,0.001885888,0.002581211],"category_scores_gemma":[0.05730098,0.0005448626,0.0007847232,0.0006801296,0.00181949,0.002872811,0.002192021,0.002783955,0.0003629615],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002899129,"about_ca_system_score_gemma":0.002432957,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005551947,"about_ca_topic_score_gemma":0.007413614,"domain_scores_codex":[0.9947782,0.003316406,0.0002201201,0.0007134797,0.0007394963,0.0002323075],"domain_scores_gemma":[0.9562903,0.03500842,0.001808025,0.004140107,0.002129419,0.0006238382],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000755994,0.0006758746,0.006052533,0.0005800219,0.0002106802,0.00006529383,0.0001470859,0.8730435,0.001795221,0.01145065,0.002283964,0.1029391],"study_design_scores_gemma":[0.00006898325,0.000173018,0.0003102552,0.00002650719,0.00001732461,0.00001057633,0.00001966938,0.9903567,0.001729669,0.006944061,0.0003359101,0.000007233214],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4046104,0.002763574,0.5766692,0.001702476,0.0002062433,0.0007020625,0.000924543,0.004923203,0.007498415],"genre_scores_gemma":[0.8406093,0.000308752,0.1564946,0.000251923,0.00003045597,0.0003246753,0.0009204162,0.000203308,0.0008566358],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01199066,"threshold_uncertainty_score":0.06341338,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09601638270821015,"score_gpt":0.2466277321237235,"score_spread":0.1506113494155134,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}