{"id":"W4311691732","doi":"10.1101/2022.12.15.22280619","title":"The Challenge Dataset – simple evaluation for safe, transparent healthcare AI deployment","year":2022,"lang":"en","type":"preprint","venue":"medRxiv","topic":"COVID-19 diagnosis using AI","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; York University; Trillium Health Centre; University of Toronto","funders":"","keywords":"Generalizability theory; Software deployment; Computer science; Artificial intelligence; Audit; Enhanced Data Rates for GSM Evolution; Generalization; Machine learning; Process (computing); Data mining; Software engineering; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.006857868,0.001632808,0.0007314333,0.001631054,0.0006326613,0.001943822,0.002735924,0.001962105,0.003466107],"category_scores_gemma":[0.02836648,0.00032962,0.001040577,0.0009985033,0.0008317434,0.002124856,0.002443194,0.001316017,0.003568552],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001372575,"about_ca_system_score_gemma":0.001429415,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006093951,"about_ca_topic_score_gemma":0.008069583,"domain_scores_codex":[0.992853,0.002953979,0.0005643162,0.00140596,0.001881273,0.0003415038],"domain_scores_gemma":[0.9876757,0.00533844,0.0006412439,0.003092228,0.002493101,0.0007592632],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003095337,0.002608972,0.07216109,0.004125469,0.0009919895,0.00148623,0.0007191478,0.08050865,0.02965286,0.004899139,0.5489582,0.250793],"study_design_scores_gemma":[0.001271614,0.004309508,0.1277272,0.001463852,0.0004373982,0.005740333,0.001533204,0.4921841,0.07122426,0.01077071,0.2829668,0.000371217],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5869858,0.00598727,0.1300058,0.008847933,0.003120235,0.005114248,0.1647213,0.05779438,0.03742302],"genre_scores_gemma":[0.6139719,0.0007563879,0.1253554,0.002663385,0.0003901515,0.001636432,0.2485778,0.002040099,0.004608336],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9931421,"threshold_uncertainty_score":0.03626835,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1930494718224837,"score_gpt":0.4524893938227347,"score_spread":0.259439922000251,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}