{"id":"W4392947622","doi":"10.1136/bmjqs-2023-016695","title":"Crowdsourcing a diagnosis? Exploring the accuracy of the size and type of group diagnosis: an experimental study","year":2024,"lang":"en","type":"article","venue":"BMJ Quality & Safety","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"McMaster University","funders":"Royal College of Physicians and Surgeons of Canada","keywords":"Medical diagnosis; Medicine; Differential diagnosis; Nominal group technique; Crowdsourcing; Group (periodic table); Diagnostic accuracy; Medical physics; Radiology; Artificial intelligence; Pathology; Computer science","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0027655,0.0001701309,0.0004645319,0.00002683917,0.0001236279,0.00003404381,0.000196735,0.00005675641,0.00009416226],"category_scores_gemma":[0.05539755,0.00009364016,0.0001755602,0.0003297267,0.0002293972,0.0001305058,0.0002259941,0.0002721174,0.000004112281],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004319766,"about_ca_system_score_gemma":0.0001221593,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009118948,"about_ca_topic_score_gemma":0.00006400756,"domain_scores_codex":[0.9974472,0.0005225746,0.001003071,0.000348233,0.0004812093,0.0001977332],"domain_scores_gemma":[0.9552547,0.04366554,0.0001873203,0.0006841107,0.00008375673,0.0001245234],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0006460752,0.003295447,0.9528488,0.0002390664,0.0002563375,0.00002708832,0.02115879,0.00001417245,0.001239591,0.001264284,0.0005675661,0.01844282],"study_design_scores_gemma":[0.001256969,0.001105674,0.9682068,0.002893746,0.0002625644,0.00001056923,0.02112624,0.0000407381,0.004334501,0.0001320342,0.0004888403,0.0001412832],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9933345,0.002620689,0.000008408475,0.002196211,0.0005253396,0.001110182,0.00002081606,0.00005176042,0.0001320798],"genre_scores_gemma":[0.998513,0.0004981821,0.00007014038,0.0002842388,0.0002322854,0.0003442613,0.000004136313,0.00002357358,0.00003022201],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.05263205,"threshold_uncertainty_score":0.9525592,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1647990118473574,"score_gpt":0.4631113570271014,"score_spread":0.2983123451797439,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}