{"id":"W1994298865","doi":"10.1371/journal.pone.0052221","title":"Assessing Diagnostic Tests: How to Correct for the Combined Effects of Interpretation and Reference Standard","year":2012,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"SUNY Downstate Medical Center; National Institutes of Health; National Institute of Neurological Disorders and Stroke; York University; State University of New York","keywords":"Sensitivity (control systems); Reliability (semiconductor); Diagnostic accuracy; Gold standard (test); Reference values; Standard deviation; Computer science; Interpreter; Diagnostic test; Statistics; Standard error; Calibration; Interpretation (philosophy); Medical physics; Medicine; Mathematics; Radiology; Internal medicine; Pediatrics; Physics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.00294141,0.0000738135,0.0002054094,0.00005100498,0.0001050541,0.0002050014,0.0002219319,0.00002657066,0.0000168374],"category_scores_gemma":[0.04526049,0.00004230416,0.00002673327,0.0001820083,0.00005390645,0.0003361491,0.00006330241,0.0000607954,0.000007218643],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003187656,"about_ca_system_score_gemma":0.00001942179,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000004195913,"about_ca_topic_score_gemma":0.000009464347,"domain_scores_codex":[0.9983552,0.0001607317,0.0002315159,0.0001654277,0.0009361248,0.0001510481],"domain_scores_gemma":[0.9791431,0.0200002,0.0001365106,0.0002794476,0.0003719126,0.00006879656],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0006421738,0.002696162,0.5424988,0.0008151719,0.0002999168,6.513993e-7,0.007107463,0.00006844896,0.1516077,0.001376362,0.006582953,0.2863043],"study_design_scores_gemma":[0.001542507,0.002569844,0.7398353,0.002041379,0.0004243618,4.975342e-7,0.002641626,0.003456729,0.2367987,0.009453291,0.0008968589,0.0003389217],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9627332,0.0003549946,0.03380107,0.001830343,0.000184107,0.0009431335,0.000007334369,0.00001055737,0.0001352714],"genre_scores_gemma":[0.9964541,0.0000229189,0.003188555,0.0001411028,0.00003663719,0.0000981019,9.094638e-7,0.00000396107,0.00005376286],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2859654,"threshold_uncertainty_score":0.9627817,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2178926583735101,"score_gpt":0.3706645710522915,"score_spread":0.1527719126787814,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}