{"id":"W6891640368","doi":"10.48448/455f-ep51","title":"How is BERT surprised? Layerwise detection of linguistic anomalies","year":2021,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Grammaticality; Language model; Security token; Transformer; Judgement; Gaussian; Word (group theory)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0007149981,0.0004848089,0.0006502293,0.001591229,0.0001986316,0.0002972715,0.0009998729,0.0003126085,0.00132805],"category_scores_gemma":[0.001835306,0.0004729523,0.0001441048,0.002769011,0.002065247,0.0001375974,0.0003285115,0.000360995,0.0002847571],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003058977,"about_ca_system_score_gemma":0.001104832,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008281817,"about_ca_topic_score_gemma":0.002435299,"domain_scores_codex":[0.9962308,0.00008665188,0.0004035387,0.001114025,0.001491539,0.0006733887],"domain_scores_gemma":[0.9970286,0.0001116803,0.0007288787,0.001160616,0.0007460938,0.0002241413],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00008908401,0.001185347,0.003101097,0.001658345,0.0004774769,0.0002711363,0.004825865,0.00006880316,0.8075925,0.002916131,0.137567,0.04024726],"study_design_scores_gemma":[0.001094005,0.0003535382,0.0005601278,0.00136698,0.0002933295,0.00008131671,0.001139815,0.008791038,0.1570671,0.001250279,0.8261385,0.001863936],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.00931463,0.01554595,0.008255919,0.0005943567,0.007667474,0.002100437,0.001547203,0.00186988,0.9531041],"genre_scores_gemma":[0.5652515,0.0001129409,0.006995596,0.0001122942,0.0008895318,0.00002008764,0.00008161457,0.0008123133,0.4257241],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.6885715,"threshold_uncertainty_score":0.9997722,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02351835849607068,"score_gpt":0.2815819742120373,"score_spread":0.2580636157159666,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}