{"id":"W6947722435","doi":"10.48448/axzg-7v02","title":"One Wug, Two Wug+s Transformer Inflection Models Hallucinate Affixes","year":2022,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Inflection; Transformer; Hallucinating; Reduplication; Hidden Markov model; Feature (linguistics); Pattern recognition (psychology); Inflection point","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001245217,0.0009814589,0.0004265369,0.0006829267,0.0004745699,0.001492875,0.001403146,0.0007054479,0.00539631],"category_scores_gemma":[0.003483252,0.0003169242,0.0009529854,0.0007511041,0.0008011635,0.001644122,0.001472576,0.001202558,0.002198393],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001212653,"about_ca_system_score_gemma":0.001899071,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0376193,"about_ca_topic_score_gemma":0.08356071,"domain_scores_codex":[0.9995989,0.00007404135,0.00002390226,0.0001387348,0.00009685561,0.00006748032],"domain_scores_gemma":[0.9989848,0.0003061587,0.00004118171,0.0003641853,0.0002428564,0.00006081714],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008923688,0.0002434223,0.03882683,0.0002851135,0.0003321108,0.000445244,0.001110366,0.2161595,0.03036084,0.01240523,0.01301803,0.685921],"study_design_scores_gemma":[0.00004177501,0.0002387413,0.0106581,0.00004183469,0.0001257212,0.0002479753,0.000516757,0.9325539,0.03041296,0.01014374,0.01495293,0.00006559884],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"other","genre_scores_codex":[0.5714632,0.0005273599,0.3769881,0.001353815,0.0002072072,0.0003772658,0.003537459,0.01837874,0.0271669],"genre_scores_gemma":[0.8782865,0.0001723223,0.1046661,0.0002458727,0.00001652903,0.0001059334,0.003903943,0.0006328651,0.01196995],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.0376193,"threshold_uncertainty_score":0.07480067,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2190072877094653,"score_gpt":0.4263552269300338,"score_spread":0.2073479392205685,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}