{"id":"W3158254122","doi":"10.3389/frai.2021.648543","title":"Considering Performance in the Automated and Manual Coding of Sociolinguistic Variables: Lessons From Variable (ING)","year":2021,"lang":"en","type":"article","venue":"Frontiers in Artificial Intelligence","topic":"Linguistic Variation and Morphology","field":"Social Sciences","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Science Foundation","keywords":"Coding (social sciences); Computer science; Natural language processing; Artificial intelligence; Speech recognition; Statistics; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001443228,0.00009386759,0.0002371935,0.00009848659,0.0002291752,0.00006453779,0.0002144218,0.0001271307,0.0001052021],"category_scores_gemma":[0.00367053,0.00009232402,0.00002210891,0.0004704371,0.0003622702,0.00006227488,0.00005371468,0.0002131914,0.000004221478],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008305792,"about_ca_system_score_gemma":0.0003549844,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002669846,"about_ca_topic_score_gemma":0.001246387,"domain_scores_codex":[0.9984297,0.0003682764,0.0004470711,0.0002593545,0.0002023585,0.0002933058],"domain_scores_gemma":[0.998946,0.0006127615,0.0001186045,0.0001547339,0.0001266969,0.00004118482],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00004334659,0.0001709194,0.03956323,0.00005371455,0.0000335059,0.0001019803,0.1089258,0.002482062,0.0007800838,0.840704,0.0005059926,0.006635319],"study_design_scores_gemma":[0.0003063132,0.00006533729,0.01288488,0.0006135816,0.0000916552,0.00001215279,0.1980486,0.2796139,0.007911367,0.4858729,0.01388184,0.0006975131],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6289379,0.001858366,0.3017766,0.004026057,0.01035654,0.000865556,0.00008801743,0.0002810631,0.05180995],"genre_scores_gemma":[0.9771043,0.0003222404,0.02213769,0.0002010913,0.0001591281,0.00001009112,0.000008076424,0.000006373154,0.0000510036],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3548312,"threshold_uncertainty_score":0.4394233,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06119933609840694,"score_gpt":0.3479488632275913,"score_spread":0.2867495271291843,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}