{"id":"W7108212453","doi":"10.17605/osf.io/qe5pw","title":"Estimating the Prevalence of Invalid PID-5 Protocols and their Impact on Scale Scores, Correlates, and Reliability","year":2023,"lang":"","type":"other","venue":"Open Science Framework","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Reliability (semiconductor); Scale (ratio); Sample (material); Validity; Estimation; Psychometrics; Criterion validity; Test (biology)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","sts","scholarly_communication","open_science","research_integrity","insufficient_payload"],"consensus_categories":["metaepi_narrow","sts","open_science","insufficient_payload"],"category_scores_codex":[0.02648978,0.001710792,0.002122972,0.0006481381,0.002421865,0.004171551,0.01172534,0.001064072,0.002006713],"category_scores_gemma":[0.0233789,0.000961501,0.0002610658,0.007024675,0.0216755,0.0020103,0.008684656,0.003280792,0.0008732571],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007406465,"about_ca_system_score_gemma":0.002966629,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001272076,"about_ca_topic_score_gemma":0.0001025141,"domain_scores_codex":[0.9877967,0.001379984,0.001838676,0.004343975,0.002599055,0.002041615],"domain_scores_gemma":[0.9823489,0.006321808,0.003231274,0.006505675,0.000641907,0.000950471],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002977079,0.001969524,0.8345675,0.00829162,0.0002817042,0.00002872723,0.02198835,0.01610433,0.002136822,0.002870891,0.008993011,0.09979039],"study_design_scores_gemma":[0.002201165,0.006419625,0.6126679,0.158945,0.0004323696,0.0001742697,0.001122867,0.146445,0.002212388,0.06195634,0.003121496,0.00430165],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.581287,0.001613407,0.002212103,0.001055144,0.002585536,0.3656302,0.004378254,0.0007050182,0.04053332],"genre_scores_gemma":[0.9065382,0.0003930111,0.04091282,0.000281397,0.0005181035,0.03021523,0.000006994915,0.002496861,0.01863731],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.335415,"threshold_uncertainty_score":0.9999047,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0379360010026904,"score_gpt":0.3797771370524428,"score_spread":0.3418411360497524,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}