{"id":"W3093967600","doi":"10.48550/arxiv.2010.09890","title":"Watch-And-Help: A Challenge for Social Perception and Human-AI Collaboration","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Reinforcement Learning in Robotics","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Perception; Psychology; Data science; Sociology; Computer science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005441078,0.001204958,0.000954688,0.0004857578,0.001225068,0.002135195,0.002005148,0.002975815,0.003410958],"category_scores_gemma":[0.02837352,0.0003947513,0.0007297769,0.0003854893,0.002677854,0.004755128,0.00465973,0.002720888,0.001303611],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001229463,"about_ca_system_score_gemma":0.00141602,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004186876,"about_ca_topic_score_gemma":0.00451886,"domain_scores_codex":[0.9919953,0.005272631,0.0002591949,0.00106523,0.00113012,0.0002774165],"domain_scores_gemma":[0.9811127,0.01142607,0.0009958366,0.003250043,0.001278986,0.0019365],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003103006,0.005111659,0.02986568,0.003453208,0.0006776251,0.0007959211,0.005651787,0.2842775,0.02618096,0.05767127,0.09276009,0.4904513],"study_design_scores_gemma":[0.0005086247,0.00355158,0.01779233,0.0003266389,0.0001066393,0.0008079535,0.004416951,0.7127702,0.02170512,0.1528064,0.08490763,0.0002998162],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5701495,0.004859583,0.3667884,0.01350036,0.00138251,0.00150746,0.003721454,0.005750624,0.03234017],"genre_scores_gemma":[0.8659854,0.0005885331,0.1240415,0.001234984,0.0001501643,0.0007333893,0.003056054,0.0003494195,0.00386053],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005441078,"threshold_uncertainty_score":0.02877557,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1105636904725638,"score_gpt":0.2406638957461023,"score_spread":0.1301002052735385,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}