{"id":"W4396636615","doi":"10.2196/50437","title":"Considerations for Quality Control Monitoring of Machine Learning Models in Clinical Practice","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institute of Biomedical Imaging and Bioengineering","keywords":"Quality (philosophy); Computer science; Control (management); Medicine; Machine learning; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3030981,0.001370673,0.002133468,0.003774759,0.002259826,0.01926749,0.007846054,0.003896692,0.003540424],"category_scores_gemma":[0.5792589,0.001385664,0.001996241,0.004907119,0.005015914,0.01435263,0.006327716,0.007543184,0.001928071],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.007744697,"about_ca_system_score_gemma":0.0248775,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01610451,"about_ca_topic_score_gemma":0.0113089,"domain_scores_codex":[0.7067736,0.2042708,0.02108161,0.009524592,0.05481709,0.003532324],"domain_scores_gemma":[0.3449593,0.4216034,0.04170138,0.0722785,0.1121163,0.007341132],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001800484,0.0008192432,0.05614122,0.002854998,0.0005630474,0.0005022425,0.007929397,0.03669494,0.00638319,0.06128516,0.07449648,0.7505295],"study_design_scores_gemma":[0.001713062,0.005743989,0.08845553,0.01221465,0.0009939196,0.001949797,0.007487448,0.2454731,0.02873264,0.1708726,0.435395,0.0009681826],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03864804,0.009088154,0.6192677,0.2882845,0.002660259,0.003641688,0.001113748,0.01060066,0.02669528],"genre_scores_gemma":[0.2950261,0.002789229,0.6783761,0.01426305,0.001589816,0.003211507,0.0009941604,0.001746809,0.002003169],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.6969019,"threshold_uncertainty_score":0.8594041,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1212489166717202,"score_gpt":0.4716219799815563,"score_spread":0.3503730633098361,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}