{"id":"W6965264423","doi":"10.3389/frai.2021.648543.s004","title":"Data_Sheet_4_Considering Performance in the Automated and Manual Coding of Sociolinguistic Variables: Lessons From Variable (ING).ZIP","year":2021,"lang":"en","type":"dataset","venue":"Figshare","topic":"Forest Ecology and Biodiversity Studies","field":"Agricultural and Biological Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Coding (social sciences); Workflow; Pronunciation; Variable (mathematics); Variation (astronomy); Natural language; Random forest; Ground truth","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03428568,0.0007946861,0.00101594,0.00432714,0.001641646,0.003335158,0.003538187,0.002018353,0.2485383],"category_scores_gemma":[0.1795201,0.0008844411,0.001389496,0.006449853,0.001192197,0.003760641,0.003516768,0.002135878,0.09166131],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002824923,"about_ca_system_score_gemma":0.005651025,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0136927,"about_ca_topic_score_gemma":0.02067887,"domain_scores_codex":[0.9764049,0.009965057,0.003523088,0.001691146,0.00767288,0.0007429803],"domain_scores_gemma":[0.7476164,0.1426912,0.01007655,0.03726604,0.05996428,0.002385594],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0009736731,0.000158854,0.009688141,0.001949686,0.00006512673,0.00008336409,0.0006742485,0.0009447934,0.0007938193,0.007291891,0.8563681,0.1210081],"study_design_scores_gemma":[0.0003992923,0.0002919112,0.05924532,0.002400567,0.00005127427,0.0001477576,0.001179461,0.002343325,0.003309763,0.008287059,0.9221497,0.0001944886],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.005926753,0.0003095757,0.01570834,0.006535792,0.0007215638,0.00387854,0.912308,0.003821219,0.05079025],"genre_scores_gemma":[0.05078622,0.0009488275,0.1333767,0.008536961,0.0006160188,0.0343853,0.7071446,0.005227196,0.05897812],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.2485383,"threshold_uncertainty_score":0.8314434,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06069834652553158,"score_gpt":0.2613075618435807,"score_spread":0.2006092153180491,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}