{"id":"W2105385152","doi":"10.1002/sim.5676","title":"Comparing diagnostic tests: trials in people with discordant test results","year":2012,"lang":"en","type":"article","venue":"Statistics in Medicine","topic":"Statistical Methods in Clinical Trials","field":"Mathematics","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Women's Health Research Institute","funders":"National Institutes of Health; Universiteit van Amsterdam; National Institute for Health and Care Research","keywords":"Statistics; Test (biology); Gold standard (test); Statistical hypothesis testing; Computer science; Medicine; Econometrics; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02772352,0.0003616108,0.00262076,0.0002759898,0.00005269805,0.00002074036,0.0003204855,0.00008895268,0.0002265384],"category_scores_gemma":[0.8694941,0.0002403867,0.00004249174,0.0007573203,0.0004475876,0.00008929633,0.0001111798,0.0007275244,0.00002779992],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001958976,"about_ca_system_score_gemma":0.00008509805,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003121401,"about_ca_topic_score_gemma":0.001934557,"domain_scores_codex":[0.9922604,0.002184955,0.003488561,0.0004613116,0.0007960029,0.0008087921],"domain_scores_gemma":[0.4581117,0.5401794,0.0007510307,0.000562416,0.0001074049,0.0002880096],"domain_codex":null,"domain_gemma":"methods","domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0009095863,0.001601089,0.6677448,0.0005865545,0.00006898374,0.0003968465,0.002384065,0.00002426902,0.00007994164,0.3017677,0.01864542,0.005790741],"study_design_scores_gemma":[0.009878109,0.0008270821,0.3379191,0.00217841,0.0002523648,0.0000320847,0.0007287342,0.0006097908,0.00005309327,0.6468367,0.0002633161,0.0004211772],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.05335245,0.000475172,0.9246786,0.002434916,0.002728372,0.00347197,0.002423014,0.0001699071,0.01026555],"genre_scores_gemma":[0.4167245,0.000117753,0.5822122,0.0001118388,0.0005787688,0.0001009306,0.0000294009,0.00004448378,0.00008006794],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8417706,"threshold_uncertainty_score":0.9802684,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5460293344991152,"score_gpt":0.5771006422312617,"score_spread":0.0310713077321465,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}