{"id":"W4311691732","doi":"10.1101/2022.12.15.22280619","title":"The Challenge Dataset – simple evaluation for safe, transparent healthcare AI deployment","year":2022,"lang":"en","type":"preprint","venue":"medRxiv","topic":"COVID-19 diagnosis using AI","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; York University; Trillium Health Centre; University of Toronto","funders":"","keywords":"Generalizability theory; Software deployment; Computer science; Artificial intelligence; Audit; Enhanced Data Rates for GSM Evolution; Generalization; Machine learning; Process (computing); Data mining; Software engineering; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.003338357,0.0003717506,0.000565326,0.000137941,0.000518988,0.00005950762,0.0005295312,0.000190077,0.0006082438],"category_scores_gemma":[0.0006931931,0.0002946474,0.0002630352,0.0001345519,0.00006468075,0.000033037,0.0004439701,0.0009318625,0.00001598624],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001167867,"about_ca_system_score_gemma":0.001296301,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000603868,"about_ca_topic_score_gemma":0.001733352,"domain_scores_codex":[0.9960151,0.0004603586,0.0006915089,0.0009853735,0.001344976,0.0005026965],"domain_scores_gemma":[0.9964011,0.0008777349,0.0002865753,0.001924417,0.0002910852,0.0002190892],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001243106,0.001559009,0.01541254,0.006835843,0.0009133192,0.0000629952,0.002891236,0.008622656,0.0001138669,0.0006967778,0.839853,0.1217957],"study_design_scores_gemma":[0.001586793,0.0004244352,0.01000047,0.0002540461,0.0006823629,0.00000347687,0.0001206035,0.008331999,0.0001741102,0.002148425,0.9759929,0.0002804298],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.03221538,0.01421908,0.001012127,0.9250185,0.002898521,0.01104326,0.01332888,0.0002244967,0.00003975944],"genre_scores_gemma":[0.866805,0.006280587,0.0003007783,0.04739848,0.001150487,0.01592914,0.06175981,0.0002331032,0.0001425859],"genre_candidate":"commentary","genre_consensus":null,"teacher_disagreement_score":0.87762,"threshold_uncertainty_score":0.9999506,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1930494718224837,"score_gpt":0.4524893938227347,"score_spread":0.259439922000251,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}