{"id":"W2964469479","doi":"","title":"Understanding the Relation Between Maximum-Entropy Inverse Reinforcement Learning and Behaviour Cloning.","year":2019,"lang":"en","type":"article","venue":"International Conference on Learning Representations","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reinforcement learning; Principle of maximum entropy; Cloning (programming); Artificial intelligence; Computer science; Relation (database); Inverse; Entropy (arrow of time); Mathematics; Machine learning; Data mining; Physics; Thermodynamics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001423015,0.0003764164,0.0005857456,0.0004956975,0.0004130627,0.001466989,0.001509281,0.001372467,0.003745248],"category_scores_gemma":[0.01488439,0.0005112981,0.0005555482,0.0003766083,0.002431075,0.00373569,0.00158099,0.002203366,0.0003348285],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001088368,"about_ca_system_score_gemma":0.0008043373,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002052657,"about_ca_topic_score_gemma":0.001838586,"domain_scores_codex":[0.9992329,0.0003098939,0.00004043762,0.0001757301,0.0001566294,0.00008449442],"domain_scores_gemma":[0.9930681,0.004860662,0.0007528689,0.0006064227,0.0004369791,0.0002749385],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001222992,0.0001241358,0.002878579,0.000169793,0.00009196407,0.0001462829,0.0003235787,0.272945,0.004688954,0.6668865,0.001304212,0.0503188],"study_design_scores_gemma":[0.00000820924,0.00003634731,0.0006175287,0.00001546159,0.00001051116,0.00004449149,0.00002320916,0.5136353,0.0006543813,0.4844929,0.000450441,0.00001136844],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.053229,0.0006498378,0.9341862,0.001338179,0.00007746276,0.00003841288,0.00006988194,0.0001559196,0.01025504],"genre_scores_gemma":[0.9333039,0.0003341891,0.06209532,0.0002185119,0.00005084621,0.00007349517,0.00008371508,0.00006023264,0.003779752],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003745248,"threshold_uncertainty_score":0.01252913,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0945946462711303,"score_gpt":0.3266313768676395,"score_spread":0.2320367305965092,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}