{"id":"W2988571156","doi":"10.1145/3359178","title":"Understanding Expert Disagreement in Medical Data Analysis through Structured Adjudication","year":2019,"lang":"en","type":"article","venue":"Proceedings of the ACM on Human-Computer Interaction","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":52,"is_retracted":false,"has_abstract":true,"ca_institutions":"Health Sciences Centre; Sunnybrook Health Science Centre; University of Toronto; University of Waterloo","funders":"Canadian Institutes of Health Research; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada","keywords":"Adjudication; CLARITY; Context (archaeology); Psychology; Presentation (obstetrics); Computer science; Data science; Medicine; Political science; Law","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.535892,0.001727443,0.002513295,0.01235789,0.008052963,0.01280543,0.006123214,0.006327357,0.002483009],"category_scores_gemma":[0.7537192,0.002047113,0.002821898,0.005687004,0.01713062,0.01372,0.01724676,0.01183781,0.0007501506],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006700978,"about_ca_system_score_gemma":0.01890315,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002898911,"about_ca_topic_score_gemma":0.00507788,"domain_scores_codex":[0.2601884,0.6441562,0.03966301,0.02000476,0.03308274,0.002904821],"domain_scores_gemma":[0.08327685,0.82098,0.0351383,0.03398255,0.02494378,0.001678528],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.001401717,0.0005892849,0.04472542,0.004342317,0.001460602,0.002166477,0.2072149,0.03272107,0.008472858,0.2738987,0.01338453,0.4096223],"study_design_scores_gemma":[0.0003234915,0.0003202403,0.005758702,0.002614259,0.0002061016,0.0005434556,0.01668498,0.09190086,0.004726007,0.8520768,0.02450229,0.0003428113],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05610432,0.001175829,0.9202162,0.01403486,0.00054165,0.002467223,0.000282949,0.0003691234,0.004807922],"genre_scores_gemma":[0.3816371,0.0004221981,0.6100379,0.003137992,0.000504079,0.003034699,0.0003737758,0.0001542566,0.0006980753],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.464108,"threshold_uncertainty_score":0.5723279,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2065316819450487,"score_gpt":0.4112688192082902,"score_spread":0.2047371372632415,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}