{"id":"W1955533988","doi":"10.6000/1929-6029.2015.04.02.2","title":"On the Relationship between the Reliability and Accuracy of Bio-Behavioral Diagnoses: Simple Math to the Rescue","year":2015,"lang":"en","type":"article","venue":"International Journal of Statistics in Medical Research","topic":"Reliability and Agreement in Measurement","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Medical diagnosis; Kappa; Mathematics; Statistics; Cohen's kappa; Equivalence (formal languages); Statistic; Youden's J statistic; Combinatorics; Psychology; Medicine; Discrete mathematics; Pathology; Receiver operating characteristic; Geometry","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.07744791,0.00007961009,0.0001933036,0.0002490095,0.0001561356,0.0002183794,0.002675238,0.00006541773,0.0002144793],"category_scores_gemma":[0.4724514,0.00003116069,0.00005094293,0.000563599,0.0007482238,0.0001137641,0.0004596939,0.001225659,0.00003814941],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001918248,"about_ca_system_score_gemma":0.0006064412,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002996633,"about_ca_topic_score_gemma":0.0002785052,"domain_scores_codex":[0.9836411,0.002410036,0.001214658,0.0001942905,0.01232827,0.0002116204],"domain_scores_gemma":[0.8517298,0.1424266,0.0003795986,0.0006181903,0.004511993,0.0003338588],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003722893,0.0003733751,0.6758875,0.000006430251,0.00002874803,0.00005750653,0.003074527,0.0007601113,0.000005147205,0.1306989,0.1194575,0.06927798],"study_design_scores_gemma":[0.0003941548,0.0004129487,0.4713234,0.0001244257,0.000005614508,0.000006858912,0.002505464,0.0008425349,0.00002181154,0.5121834,0.01213873,0.0000407284],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.901176,0.00005546113,0.007506433,0.09035337,0.0003416706,0.0002913848,0.00009355734,0.000001161126,0.0001809158],"genre_scores_gemma":[0.9986672,0.0000590105,0.0006203376,0.0003689583,0.0002234036,0.0000117003,0.000002031544,0.000004049288,0.00004333381],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3950035,"threshold_uncertainty_score":0.9499615,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5558141781385264,"score_gpt":0.5812588006269411,"score_spread":0.02544462248841473,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}