{"id":"W3021643468","doi":"10.18653/v1/2020.acl-main.679","title":"The Sensitivity of Language Models and Humans to Winograd Schema Perturbations","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Schema (genetic algorithms); Language model; Associative property; Artificial intelligence; Task (project management); Language understanding; Cognitive psychology; Synonym (taxonomy); Natural language processing; Machine learning; Psychology; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003189097,0.000115917,0.0001660252,0.00003840917,0.00009986722,0.0001372753,0.0004413313,0.00006712109,0.00000141355],"category_scores_gemma":[0.00007231109,0.00008420043,0.00005543604,0.00008539574,0.00003057637,0.00009176276,0.001916495,0.0002167565,0.000002778369],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001482564,"about_ca_system_score_gemma":0.00006210244,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003289844,"about_ca_topic_score_gemma":0.0001618235,"domain_scores_codex":[0.9990318,0.00007530783,0.0001798497,0.0003901598,0.0001949366,0.0001278976],"domain_scores_gemma":[0.9988626,0.0001703176,0.00006070197,0.0007600983,0.00006914383,0.00007716305],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000003991421,0.00003492673,0.0001053258,0.0001323519,0.00007462483,0.00001785695,0.03363614,0.0467747,0.004770339,0.8460972,0.0009226688,0.06742989],"study_design_scores_gemma":[0.00003487368,0.000007792101,0.000190476,0.00002547135,0.000005286975,0.000001941583,0.0001782493,0.9816765,0.0006281995,0.01703911,0.0001078342,0.0001043362],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04815917,0.0001462744,0.9402245,0.008371902,0.0001165964,0.0002093473,0.000005237279,0.00009876076,0.002668245],"genre_scores_gemma":[0.8929317,0.00001382446,0.1063216,0.0003609441,0.00005150556,0.000009813089,0.00000150582,0.000005226606,0.0003039306],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9349017,"threshold_uncertainty_score":0.3433594,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05336245230994675,"score_gpt":0.2853620135267133,"score_spread":0.2319995612167665,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}