{"id":"W1994298865","doi":"10.1371/journal.pone.0052221","title":"Assessing Diagnostic Tests: How to Correct for the Combined Effects of Interpretation and Reference Standard","year":2012,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"SUNY Downstate Medical Center; National Institutes of Health; National Institute of Neurological Disorders and Stroke; York University; State University of New York","keywords":"Sensitivity (control systems); Reliability (semiconductor); Diagnostic accuracy; Gold standard (test); Reference values; Standard deviation; Computer science; Interpreter; Diagnostic test; Statistics; Standard error; Calibration; Interpretation (philosophy); Medical physics; Medicine; Mathematics; Radiology; Internal medicine; Pediatrics; Physics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.09031046,0.0025195,0.00428048,0.005042796,0.001303573,0.004342535,0.004304928,0.00555703,0.002607812],"category_scores_gemma":[0.3758319,0.001800897,0.002488394,0.003739163,0.004800959,0.007539719,0.004581641,0.004378718,0.002498024],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001597126,"about_ca_system_score_gemma":0.005220817,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006426156,"about_ca_topic_score_gemma":0.005418115,"domain_scores_codex":[0.942035,0.03576608,0.004121707,0.004957615,0.01256336,0.0005562202],"domain_scores_gemma":[0.8013542,0.1455422,0.0111018,0.01774858,0.02321382,0.00103943],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002550705,0.0001847595,0.0205045,0.001712361,0.00144605,0.000345238,0.001604301,0.0526007,0.00579018,0.03665238,0.0110907,0.8678138],"study_design_scores_gemma":[0.0003047021,0.0008570163,0.02690028,0.001842246,0.001234086,0.00325227,0.001089175,0.4416645,0.02497848,0.4627591,0.03410146,0.001016663],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.003432847,0.0007350951,0.9926682,0.001684375,0.0001595181,0.000186421,0.00007363794,0.0005619512,0.0004980106],"genre_scores_gemma":[0.03058394,0.000491105,0.9677565,0.000289684,0.00009525281,0.0002951736,0.00006672276,0.0001446437,0.00027706],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9096895,"threshold_uncertainty_score":0.4776131,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2178926583735101,"score_gpt":0.3706645710522915,"score_spread":0.1527719126787814,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}