{"id":"W3021643468","doi":"10.18653/v1/2020.acl-main.679","title":"The Sensitivity of Language Models and Humans to Winograd Schema Perturbations","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Schema (genetic algorithms); Language model; Associative property; Artificial intelligence; Task (project management); Language understanding; Cognitive psychology; Synonym (taxonomy); Natural language processing; Machine learning; Psychology; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01388238,0.002698257,0.001464445,0.00150489,0.001020195,0.004422647,0.002706924,0.002545243,0.005327607],"category_scores_gemma":[0.06111458,0.0009791972,0.001393595,0.001413006,0.002062144,0.008459159,0.004059029,0.006842679,0.004116045],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001631232,"about_ca_system_score_gemma":0.001449027,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007368538,"about_ca_topic_score_gemma":0.010161,"domain_scores_codex":[0.9875448,0.005615902,0.0006856447,0.004389717,0.001298709,0.0004652852],"domain_scores_gemma":[0.9633334,0.02337967,0.001343749,0.009670096,0.001382896,0.0008902313],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002869924,0.0009006233,0.0723984,0.002408477,0.002186702,0.0009081684,0.002211591,0.1797511,0.02243228,0.01575278,0.1634914,0.5346885],"study_design_scores_gemma":[0.0004146509,0.0005549341,0.02556248,0.0004904334,0.0003614674,0.001498703,0.001820481,0.7735078,0.02613694,0.1181124,0.05124385,0.0002958952],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7068186,0.01534856,0.192064,0.01193469,0.002589307,0.0005040356,0.01944708,0.02307838,0.02821525],"genre_scores_gemma":[0.9064029,0.001215418,0.05444836,0.002528396,0.0002428275,0.0002561254,0.02915709,0.001373277,0.004375739],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01388238,"threshold_uncertainty_score":0.07341796,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05336245230994675,"score_gpt":0.2853620135267133,"score_spread":0.2319995612167665,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}