{"id":"W2971569798","doi":"10.18653/v1/d19-1006","title":"How Contextual are Contextualized Word Representations? Comparing the Geometry of BERT, ELMo, and GPT-2 Embeddings","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":60,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Word (group theory); Similarity (geometry); Context (archaeology); Computer science; Embedding; Cosine similarity; Natural language processing; Task (project management); Artificial intelligence; Variance (accounting); Linguistics; Pattern recognition (psychology); Image (mathematics); History; Philosophy","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007347153,0.0007911428,0.0005750758,0.0008034671,0.0004279216,0.001512001,0.0008161032,0.001062917,0.003666861],"category_scores_gemma":[0.007029667,0.0005585736,0.0007164,0.0007825188,0.001236546,0.005083159,0.00169571,0.001848831,0.00090052],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009588184,"about_ca_system_score_gemma":0.0006941722,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00530776,"about_ca_topic_score_gemma":0.007069892,"domain_scores_codex":[0.9995824,0.0001393154,0.00002059778,0.000134099,0.00006027145,0.00006329345],"domain_scores_gemma":[0.9988006,0.0005185638,0.0001040379,0.0002850853,0.0001967956,0.00009484975],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007066781,0.0000840931,0.007229445,0.0004141162,0.0002204455,0.000370159,0.00141318,0.3626519,0.01848567,0.2955805,0.00998347,0.3028603],"study_design_scores_gemma":[0.00003858819,0.0001107644,0.002528853,0.00005699975,0.00005662353,0.0002062583,0.0002729897,0.719627,0.003739458,0.2680043,0.005314164,0.00004398512],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2592553,0.001976175,0.7207229,0.002701844,0.0001750616,0.00007480948,0.00130036,0.001740105,0.01205349],"genre_scores_gemma":[0.8840712,0.001038303,0.1078029,0.0003387462,0.0001049053,0.0001429223,0.001911839,0.0004205991,0.004168607],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00530776,"threshold_uncertainty_score":0.01226681,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05486834225772868,"score_gpt":0.2959037959050607,"score_spread":0.241035453647332,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}