{"id":"W3197766881","doi":"10.25189/2675-4916.2021.v2.n3.id399","title":"Quantifying the Differences Between Lexical Categories","year":2021,"lang":"en","type":"article","venue":"Cadernos de Linguística","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Humber Polytechnic","funders":"","keywords":"Linguistics; Categorization; Determinative; Computer science; Natural language processing; Grammar; Pronoun; Psychology; Artificial intelligence; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000256518,0.0001417239,0.000183416,0.00003758303,0.0002948736,0.0004739901,0.001138828,0.0001240529,0.00001283471],"category_scores_gemma":[0.001048159,0.00009488535,0.0000649498,0.0003550004,0.0001395472,0.0001664807,0.0004259823,0.000477002,0.00001966883],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003550159,"about_ca_system_score_gemma":0.0002103385,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002575479,"about_ca_topic_score_gemma":0.000008182809,"domain_scores_codex":[0.9987023,0.000117731,0.0002030554,0.0003490063,0.0002682055,0.0003596653],"domain_scores_gemma":[0.9985753,0.000644241,0.00006705633,0.0004882736,0.0001328114,0.00009233213],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000004077208,0.00004966247,0.04747988,0.0001055382,0.00006266832,0.000358694,0.006473158,0.000002665403,0.004615936,0.868842,0.0008469456,0.0711588],"study_design_scores_gemma":[0.0003238343,0.0001065424,0.03552792,0.0002917175,0.0001066057,0.0003102816,0.0008053554,0.01324168,0.2950345,0.6466976,0.006560279,0.0009936402],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03717791,0.00627256,0.9476518,0.007241437,0.0002695551,0.00008181627,0.00000269758,0.0006965196,0.0006056955],"genre_scores_gemma":[0.8513389,0.00002391177,0.1476448,0.0005480652,0.0002826525,0.000008823803,0.000002346995,0.000008868394,0.0001417014],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8141609,"threshold_uncertainty_score":0.4570697,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06088552525081525,"score_gpt":0.3226802791093785,"score_spread":0.2617947538585633,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}