{"id":"W2944733665","doi":"10.1177/0894439319846622","title":"Automatic Coding of Text Answers to Open-Ended Questions: Should You Double Code the Training Data?","year":2019,"lang":"en","type":"article","venue":"Social Science Computer Review","topic":"Survey Methodology and Nonresponse","field":"Social Sciences","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Coding (social sciences); Computer science; Natural language processing; Artificial intelligence; Training set; Statistical model; Machine learning; Information retrieval; Speech recognition; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1434517,0.001564982,0.001370489,0.00457644,0.002049305,0.00264822,0.003017341,0.001899763,0.008306833],"category_scores_gemma":[0.4376726,0.001053519,0.001201955,0.005209388,0.0027587,0.005416417,0.004938513,0.002350575,0.003578127],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003864843,"about_ca_system_score_gemma":0.00678113,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003603115,"about_ca_topic_score_gemma":0.005933535,"domain_scores_codex":[0.8073882,0.1498485,0.01607505,0.01054217,0.01347413,0.002671857],"domain_scores_gemma":[0.4061963,0.3718147,0.06274667,0.07701775,0.07927883,0.002945712],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003139496,0.001145995,0.05491919,0.006080294,0.0003467916,0.0002264461,0.04509845,0.00481746,0.01928301,0.03049207,0.03673416,0.7977167],"study_design_scores_gemma":[0.002662695,0.003610352,0.1972262,0.01151266,0.0009936368,0.001355686,0.05973185,0.1189465,0.1171639,0.1813121,0.3041355,0.001348915],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2788642,0.001016384,0.6648597,0.007103278,0.001064252,0.02531219,0.004604202,0.002670742,0.01450511],"genre_scores_gemma":[0.4504234,0.0007266917,0.4899511,0.002299255,0.0002746936,0.043997,0.004352257,0.0006606717,0.007315033],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8565483,"threshold_uncertainty_score":0.7586541,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5952878803019198,"score_gpt":0.5472651412840432,"score_spread":0.04802273901787657,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}