{"id":"W7108212453","doi":"10.17605/osf.io/qe5pw","title":"Estimating the Prevalence of Invalid PID-5 Protocols and their Impact on Scale Scores, Correlates, and Reliability","year":2023,"lang":"","type":"other","venue":"Open Science Framework","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Reliability (semiconductor); Scale (ratio); Sample (material); Validity; Estimation; Psychometrics; Criterion validity; Test (biology)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.05244791,0.0004748013,0.0004721309,0.003521597,0.002458474,0.001805493,0.002140919,0.0007009469,0.001630356],"category_scores_gemma":[0.09205542,0.0004020093,0.0007368244,0.003016219,0.001563021,0.0009614363,0.001642116,0.0007738266,0.0004514429],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00939904,"about_ca_system_score_gemma":0.01249838,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.418954,"about_ca_topic_score_gemma":0.5885482,"domain_scores_codex":[0.9636943,0.01711469,0.00387813,0.001949005,0.01198304,0.001380817],"domain_scores_gemma":[0.9337623,0.01733966,0.0159598,0.006613299,0.02509991,0.001225068],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00009686829,0.0001490798,0.9717031,0.0001172998,0.00009162414,0.00004074799,0.001882627,0.0002114775,0.0003138078,0.0004505488,0.001605419,0.02333743],"study_design_scores_gemma":[0.00001402378,0.0001741963,0.9929036,0.0001225891,0.00004901868,0.0000981303,0.001586246,0.001333785,0.0007460356,0.0001691578,0.002785461,0.0000177201],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9636696,0.001044862,0.01443152,0.0008776322,0.00007718175,0.003615774,0.003506043,0.0001098547,0.01266757],"genre_scores_gemma":[0.9692756,0.000528305,0.02038641,0.0005841411,0.00003007822,0.003501783,0.003330293,0.00002791662,0.002335476],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9475521,"threshold_uncertainty_score":0.8330309,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0379360010026904,"score_gpt":0.3797771370524428,"score_spread":0.3418411360497524,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}