{"id":"W3112544537","doi":"10.12758/mda.2020.10","title":"Coding Text Answers to Open-ended Questions: Human Coders and Statistical Learning Algorithms Make Similar Mistakes","year":2021,"lang":"en","type":"article","venue":"Social Science Open Access Repository (GESIS – Leibniz Institute for the Social Sciences)","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Statistical model; Coding (social sciences); Computer science; Statistical learning; Natural language processing; Artificial intelligence; Statistical analysis; Correlation; Speech recognition; Machine learning; Statistics; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.09162764,0.001617877,0.001493023,0.004963823,0.001716583,0.003998589,0.002301441,0.002714322,0.002663319],"category_scores_gemma":[0.5347641,0.0008634264,0.001016188,0.00552159,0.004970829,0.00449371,0.004255906,0.002319305,0.00155808],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002683436,"about_ca_system_score_gemma":0.002252787,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00259075,"about_ca_topic_score_gemma":0.00246923,"domain_scores_codex":[0.7541195,0.1572917,0.01801416,0.01948674,0.04865811,0.0024299],"domain_scores_gemma":[0.2734294,0.5795603,0.05698637,0.04501395,0.04348148,0.001528515],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003636315,0.00119182,0.2468067,0.006458876,0.00314615,0.001858803,0.1488599,0.01836528,0.02829741,0.03594349,0.03786603,0.4675693],"study_design_scores_gemma":[0.0009740108,0.001480737,0.3332318,0.005762583,0.001143667,0.004009732,0.06132422,0.1705151,0.06432169,0.2620559,0.09377997,0.001400567],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5628943,0.001014754,0.4116781,0.002890873,0.000742369,0.002716201,0.002198795,0.002059523,0.0138051],"genre_scores_gemma":[0.8783931,0.0003688845,0.1094739,0.002550685,0.0001586168,0.002615233,0.002431507,0.0006830497,0.00332506],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9083723,"threshold_uncertainty_score":0.484579,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1583214395006781,"score_gpt":0.5040170616981421,"score_spread":0.345695622197464,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}