{"id":"W6964938808","doi":"10.3389/frai.2021.648543.s005","title":"Data_Sheet_5_Considering Performance in the Automated and Manual Coding of Sociolinguistic Variables: Lessons From Variable (ING).ZIP","year":2021,"lang":"en","type":"dataset","venue":"Figshare","topic":"Tree-ring climate responses","field":"Earth and Planetary Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Coding (social sciences); Workflow; Pronunciation; Variable (mathematics); Variation (astronomy); Natural language; Random forest; Ground truth","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.03136724,0.0007932123,0.001067218,0.004468912,0.001664499,0.003475822,0.003536489,0.002125909,0.2997015],"category_scores_gemma":[0.1759496,0.0009337357,0.001403552,0.006407608,0.001093423,0.003767597,0.003605843,0.002188063,0.112007],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002868132,"about_ca_system_score_gemma":0.0058307,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01260803,"about_ca_topic_score_gemma":0.01977127,"domain_scores_codex":[0.9781804,0.008628296,0.003429707,0.00159392,0.007429102,0.0007385667],"domain_scores_gemma":[0.7607898,0.1323014,0.009548293,0.03479036,0.06002699,0.002543206],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0007822979,0.0001398747,0.006351794,0.001807467,0.00005205781,0.0000671658,0.0004712296,0.0007121722,0.0005863076,0.005932753,0.8880079,0.09508909],"study_design_scores_gemma":[0.0003808309,0.0002469171,0.04334804,0.002243136,0.00004267351,0.0001265029,0.0008898553,0.001752359,0.002529921,0.007061089,0.9412187,0.0001600578],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.003893381,0.0002351337,0.01110698,0.005198286,0.0006233266,0.003242263,0.9267142,0.003365864,0.0456205],"genre_scores_gemma":[0.03736347,0.0008056182,0.1035024,0.008082182,0.0005577081,0.03147402,0.7565029,0.00458115,0.05713055],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.7002985,"threshold_uncertainty_score":0.9988908,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0570777145749975,"score_gpt":0.2873544951757112,"score_spread":0.2302767806007137,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}