{"id":"W2050084558","doi":"10.1016/j.healthpol.2004.05.004","title":"The consistency of panelists’ appropriateness ratings: do experts produce clinically logical scores for rectal cancer treatment?","year":2004,"lang":"en","type":"article","venue":"Health Policy","topic":"Colorectal Cancer Surgical Treatments","field":"Medicine","cited_by":5,"is_retracted":false,"has_abstract":false,"ca_institutions":"Institute for Clinical Evaluative Sciences; Princess Margaret Cancer Centre; University of Toronto","funders":"National Cancer Institute; Ontario Ministry of Health and Long-Term Care","keywords":"Medicine; Consistency (knowledge bases); Rating scale; Appropriateness criteria; Colorectal cancer; Cancer; Statistics; Internal medicine; Radiology; Mathematics; Artificial intelligence; Computer science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004060817,0.0002470831,0.0007112307,0.00006533747,0.0003122524,0.00002052889,0.00011791,0.0001173682,0.00002058438],"category_scores_gemma":[0.001080849,0.0001369754,0.0002466983,0.0003101446,0.0003086133,0.00003677672,0.00004005271,0.000105355,0.00000462175],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001074539,"about_ca_system_score_gemma":0.003580605,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006794726,"about_ca_topic_score_gemma":0.0004934946,"domain_scores_codex":[0.9976506,0.0001220219,0.0008554532,0.0004796894,0.0002910475,0.0006012233],"domain_scores_gemma":[0.998072,0.0005348886,0.000380598,0.0004386763,0.0002356134,0.0003382302],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.02112377,0.005778198,0.09513272,0.002016311,0.001008243,0.0000930484,0.01264934,0.0001005224,0.001531228,0.0195028,0.001732619,0.8393312],"study_design_scores_gemma":[0.1038872,0.09663856,0.5874824,0.005325655,0.0008659525,0.0008123191,0.003944466,0.0001648411,0.01834384,0.02557018,0.1547169,0.002247625],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9394859,0.005472075,0.00003758398,0.04894641,0.0004353127,0.003914118,0.00009126066,0.0001003582,0.001516967],"genre_scores_gemma":[0.9905679,0.003286433,0.001037251,0.002223602,0.0008497603,0.001518058,0.00002089863,0.00003560941,0.0004605071],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8370836,"threshold_uncertainty_score":0.9998191,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1100990112215007,"score_gpt":0.4554487002357484,"score_spread":0.3453496890142477,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}