{"id":"W4200056893","doi":"10.1109/iros51168.2021.9636479","title":"A Marginal Log-Likelihood Approach for the Estimation of Discount Factors of Multiple Experts in Inverse Reinforcement Learning","year":2021,"lang":"en","type":"article","venue":"2021 IEEE/RSJ International Conference on Intelligent Robots and Systems (IROS)","topic":"Decision-Making and Behavioral Economics","field":"Decision Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Marginal likelihood; Computer science; Latent variable; Machine learning; Probabilistic logic; Hyperparameter; Expectation–maximization algorithm; Markov decision process; Artificial intelligence; Principle of maximum entropy; Reinforcement learning; Latent variable model; Likelihood function; Variable (mathematics); Markov process; Bayesian probability; Mathematical optimization; Mathematics; Maximum likelihood; Estimation theory; Statistics; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006132788,0.001545736,0.00199941,0.001027442,0.0004676502,0.001731939,0.003211849,0.001852734,0.003315647],"category_scores_gemma":[0.02345035,0.001171969,0.001171197,0.0009347564,0.002408374,0.003218731,0.002263661,0.003446477,0.0005355173],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002118034,"about_ca_system_score_gemma":0.002137832,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003433055,"about_ca_topic_score_gemma":0.002952829,"domain_scores_codex":[0.997471,0.001519819,0.0000901028,0.0003963087,0.000362755,0.0001600064],"domain_scores_gemma":[0.9920596,0.006122071,0.0005949241,0.0003712886,0.0005736506,0.0002783523],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009388897,0.00006406177,0.0008941847,0.0001022283,0.00007071064,0.00009116927,0.0001281496,0.9327578,0.0007332251,0.03982279,0.0006939997,0.02454776],"study_design_scores_gemma":[0.000007709553,0.00002376903,0.00008501827,0.00001133304,0.000006119768,0.00001834367,0.000006715317,0.981199,0.0002069366,0.01819363,0.0002308206,0.00001058289],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.003413895,0.00008172862,0.9958747,0.0001294235,0.00000877043,0.00002931176,0.00002062922,0.00007741838,0.0003640195],"genre_scores_gemma":[0.5344641,0.0003721127,0.4603133,0.000295296,0.0001043216,0.0005144161,0.0002633934,0.0002205391,0.003452456],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006132788,"threshold_uncertainty_score":0.03243363,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2132706932633973,"score_gpt":0.3892692312640018,"score_spread":0.1759985380006045,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}