{"id":"W3176017220","doi":"10.48550/arxiv.2106.10318","title":"Sample Efficient Social Navigation Using Inverse Reinforcement Learning","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Evacuation and Crowd Dynamics","field":"Engineering","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Sample (material); Reinforcement learning; Reinforcement; Inverse; Computer science; Artificial intelligence; Psychology; Mathematics; Social psychology; Physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001195651,0.001202994,0.001473439,0.0005876272,0.0005388116,0.0007537291,0.001805107,0.001292513,0.002271129],"category_scores_gemma":[0.005647979,0.0006347428,0.0005991606,0.0003764672,0.001341439,0.001359322,0.001767728,0.001663214,0.0005401524],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001034113,"about_ca_system_score_gemma":0.001963601,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007091854,"about_ca_topic_score_gemma":0.005923298,"domain_scores_codex":[0.9993407,0.0002025952,0.00003262432,0.0001800593,0.0001562678,0.00008769516],"domain_scores_gemma":[0.9977597,0.00130093,0.0002505617,0.0002681123,0.000285336,0.0001353201],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001239391,0.0001089456,0.001442327,0.00004715638,0.00003962632,0.00007325998,0.00009869732,0.9188205,0.001172571,0.006064591,0.001211546,0.07079681],"study_design_scores_gemma":[0.00001327822,0.00002185745,0.00005299291,0.000003311728,0.000002686086,0.00001020374,0.000007037133,0.9963348,0.0003095283,0.003014076,0.0002269599,0.000003192746],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02083594,0.00009250337,0.9763868,0.0001850003,0.00003727831,0.00006773199,0.00003766557,0.001121455,0.001235543],"genre_scores_gemma":[0.7583929,0.00007196606,0.2376871,0.0002197527,0.00004511323,0.0002840186,0.0002618293,0.0001752292,0.002862122],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007091854,"threshold_uncertainty_score":0.01410115,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06372074770903365,"score_gpt":0.1950972624375552,"score_spread":0.1313765147285215,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}