{"id":"W4402008279","doi":"10.31235/osf.io/wg82k","title":"Updating “The Future of Coding”: Qualitative Coding with Generative Large Language Models","year":2024,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Generative grammar; Coding (social sciences); Computer science; Natural language processing; Generative model; Artificial intelligence; Linguistics; Mathematics; Statistics; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06481946,0.001247876,0.0007555846,0.003183166,0.003242186,0.009975852,0.003967748,0.002425028,0.009984042],"category_scores_gemma":[0.3187526,0.001336766,0.001477156,0.003146829,0.01772363,0.02127168,0.008085554,0.005869536,0.002702217],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006824589,"about_ca_system_score_gemma":0.008389481,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005791496,"about_ca_topic_score_gemma":0.006537707,"domain_scores_codex":[0.9145749,0.07312189,0.001736337,0.004605968,0.005267912,0.0006930139],"domain_scores_gemma":[0.6655261,0.2557914,0.008587294,0.05052328,0.01783624,0.001735809],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001572794,0.00005341863,0.003986105,0.0007292975,0.00007290308,0.000173471,0.05071019,0.009883768,0.001830337,0.8257025,0.01074033,0.09596028],"study_design_scores_gemma":[0.00004648574,0.00002675466,0.0005489324,0.0005809358,0.00002438122,0.000137664,0.005761484,0.04618986,0.002755214,0.9004273,0.04340364,0.00009730907],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.009987927,0.0003846325,0.9718521,0.008509685,0.0002898029,0.0002139195,0.0006269579,0.0009262388,0.007208823],"genre_scores_gemma":[0.3044109,0.0005103986,0.6858631,0.002543041,0.0001878506,0.001271919,0.0009240405,0.00142165,0.002867078],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9351805,"threshold_uncertainty_score":0.3428022,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04143052169491773,"score_gpt":0.3302597114237466,"score_spread":0.2888291897288289,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}