{"id":"W3190138084","doi":"10.1093/jssam/smab022","title":"A Model-Assisted Approach for Finding Coding Errors in Manual Coding of Open-Ended Questions","year":2021,"lang":"en","type":"article","venue":"Journal of Survey Statistics and Methodology","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"Actua; University of Waterloo; Roche (Canada)","funders":"","keywords":"Coding (social sciences); Computer science; Statistics; Natural language processing; Artificial intelligence; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.05606314,0.002557892,0.002239944,0.006392811,0.001906153,0.003326988,0.004967253,0.003189472,0.002808691],"category_scores_gemma":[0.1579796,0.001601642,0.002124382,0.003226386,0.001789765,0.003241448,0.004391115,0.004175576,0.001588715],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003281877,"about_ca_system_score_gemma":0.005248129,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009178651,"about_ca_topic_score_gemma":0.01451252,"domain_scores_codex":[0.9337932,0.0496453,0.002749562,0.006656197,0.006266864,0.0008888423],"domain_scores_gemma":[0.765063,0.1815021,0.01477125,0.0189258,0.01823824,0.001499638],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001521416,0.001233367,0.02988261,0.0007020172,0.0009402582,0.0003612837,0.004892926,0.1611781,0.00865052,0.0118042,0.01096145,0.7678719],"study_design_scores_gemma":[0.00007303025,0.0001427766,0.002178024,0.00007859848,0.00004606645,0.000118862,0.0003501175,0.9820195,0.002986599,0.01077024,0.001165308,0.00007074501],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01791999,0.0000631522,0.9778147,0.0003239305,0.00004360281,0.0003998852,0.0001374609,0.002852733,0.0004444862],"genre_scores_gemma":[0.2254403,0.00004146875,0.7714089,0.0002798273,0.00003612025,0.001233656,0.0005281527,0.000209075,0.0008225068],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9439369,"threshold_uncertainty_score":0.2964937,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7870611956524755,"score_gpt":0.5441244124235828,"score_spread":0.2429367832288927,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}