{"id":"W4406256502","doi":"10.1101/2025.01.07.631773","title":"Biased sampling confounds machine learning prediction of antimicrobial resistance","year":2025,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Bacterial Identification and Susceptibility Testing","field":"Biochemistry, Genetics and Molecular Biology","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Sampling (signal processing); Artificial intelligence; Machine learning; Resistance (ecology); Antimicrobial; Computer science; Statistics; Mathematics; Biology; Microbiology; Ecology; Computer vision","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02336038,0.0009679787,0.001607079,0.0009795635,0.0008907741,0.002348641,0.001760586,0.001943083,0.0009073158],"category_scores_gemma":[0.07372756,0.0007032244,0.001336431,0.001019724,0.002416372,0.001891122,0.001508518,0.002465333,0.0004114833],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001179837,"about_ca_system_score_gemma":0.0009012609,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004723437,"about_ca_topic_score_gemma":0.004241277,"domain_scores_codex":[0.9886557,0.008893982,0.0003764133,0.001186232,0.0006139067,0.000273726],"domain_scores_gemma":[0.9268512,0.06117487,0.003339936,0.005917148,0.002110867,0.000605997],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007981128,0.0002036249,0.2448569,0.0003115034,0.001433108,0.0003735303,0.0003907843,0.6795337,0.01018559,0.006968919,0.005039829,0.04990451],"study_design_scores_gemma":[0.00005244451,0.0001040027,0.0195702,0.00004810954,0.0001002645,0.00009477112,0.00006488148,0.958581,0.002680772,0.01782222,0.0008496545,0.00003181339],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8377792,0.002644719,0.1518204,0.003748619,0.0002054591,0.00006929567,0.0007544136,0.001155272,0.001822729],"genre_scores_gemma":[0.9866307,0.0002102751,0.01154785,0.000446965,0.0001038882,0.00003331468,0.0006942565,0.00008318934,0.0002494508],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9766396,"threshold_uncertainty_score":0.123543,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02115353240114829,"score_gpt":0.2412234600133923,"score_spread":0.220069927612244,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}