{"id":"W4416035516","doi":"10.18653/v1/2025.emnlp-main.1525","title":"Lemmatization as a Classification Task: Results from Arabic across Multiple Genres","year":2025,"lang":"","type":"article","venue":"","topic":"Second Language Acquisition and Learning","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Lemmatisation; Arabic; Training set; Set (abstract data type); Feature (linguistics)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0007719714,0.0003795402,0.000406156,0.0002544085,0.0005830577,0.0003656805,0.0004160616,0.0008723935,0.05978538],"category_scores_gemma":[0.00171136,0.0003927728,0.000185441,0.0008238313,0.0001583093,0.000235295,0.0001111627,0.0007822291,0.005906297],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001383044,"about_ca_system_score_gemma":0.0001356491,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003600641,"about_ca_topic_score_gemma":0.0003913706,"domain_scores_codex":[0.9959304,0.000791889,0.001214319,0.001173567,0.0002700436,0.0006197368],"domain_scores_gemma":[0.9967023,0.001284738,0.0005038981,0.001142721,0.0002197153,0.0001466011],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.007140456,0.002010535,0.007565794,0.0001694588,0.001606703,0.0001677912,0.1756003,0.0007659654,0.07391363,0.0525119,0.1528315,0.5257159],"study_design_scores_gemma":[0.01984174,0.0002439932,0.3532678,0.0004201987,0.0002346171,0.00001811478,0.1880599,0.04187123,0.005110358,0.001317464,0.3884306,0.001184041],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7594472,0.006108745,0.0182999,0.005120451,0.002766596,0.00080467,0.0004651276,0.000349858,0.2066374],"genre_scores_gemma":[0.923514,0.00007839342,0.0004898632,0.01587548,0.00039709,0.0000586186,0.001837389,0.00003637605,0.05771277],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5245319,"threshold_uncertainty_score":0.9998524,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03021961230895533,"score_gpt":0.3677307304812117,"score_spread":0.3375111181722564,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}