{"id":"W4213112533","doi":"10.3758/s13428-022-01801-y","title":"Machine learning to detect invalid text responses: Validation and comparison to existing detection methods","year":2022,"lang":"en","type":"article","venue":"Behavior Research Methods","topic":"Text and Document Classification Technologies","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Natural language processing; Artificial intelligence; Coding (social sciences); Word (group theory); Machine learning; Code (set theory); Quality (philosophy); Information retrieval; Linguistics; Statistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01474035,0.001282578,0.001231584,0.004876309,0.0007867283,0.001496353,0.002125896,0.001788404,0.002661082],"category_scores_gemma":[0.05607874,0.0002318615,0.0007132619,0.001836969,0.0006271202,0.002075229,0.001179228,0.00162544,0.002238751],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008603192,"about_ca_system_score_gemma":0.001589573,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002752647,"about_ca_topic_score_gemma":0.002274303,"domain_scores_codex":[0.9872499,0.006348735,0.001201872,0.001649526,0.003106789,0.0004432721],"domain_scores_gemma":[0.888906,0.08579237,0.004960884,0.006912415,0.01243318,0.0009950494],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003738154,0.00514031,0.1016574,0.001289971,0.0006364932,0.0002319225,0.0009388463,0.01870285,0.01404835,0.001685065,0.007489231,0.8444415],"study_design_scores_gemma":[0.0003346205,0.002258298,0.05206369,0.0002517815,0.0004196089,0.0007968134,0.0007058362,0.9087516,0.02681953,0.003375618,0.004108669,0.0001138669],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.6149876,0.002327274,0.3637306,0.0006438135,0.0006727094,0.00203553,0.002804031,0.00630556,0.006492964],"genre_scores_gemma":[0.8483266,0.0006065616,0.1428411,0.0002884029,0.0001951293,0.0008245981,0.002699989,0.0002733419,0.003944223],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.01474035,"threshold_uncertainty_score":0.07795537,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.421437191862916,"score_gpt":0.5856823952146699,"score_spread":0.1642452033517539,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}