{"id":"W4411299320","doi":"10.1093/labmed/lmaf013","title":"Performance metrics for machine learning solutions in laboratory medicine","year":2025,"lang":"en","type":"review","venue":"Laboratory Medicine","topic":"Clinical Laboratory Practices and Quality Control","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Implementation; Computer science; Machine learning; Artificial intelligence; Software engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01468435,0.001531017,0.003057207,0.009469652,0.0004749106,0.00319551,0.002494342,0.002063975,0.005574181],"category_scores_gemma":[0.0467896,0.0005424719,0.003076832,0.008485693,0.001077574,0.003164044,0.001315471,0.002284792,0.001731072],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002957948,"about_ca_system_score_gemma":0.003503309,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002709156,"about_ca_topic_score_gemma":0.003458163,"domain_scores_codex":[0.989437,0.0038711,0.002410783,0.0009017666,0.00317127,0.0002081044],"domain_scores_gemma":[0.9622755,0.02897018,0.003470459,0.0006141398,0.0044318,0.0002379205],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000218886,0.0001031447,0.001887753,0.1123311,0.001589062,0.00006166218,0.0001431496,0.002804202,0.00068078,0.01375743,0.0236399,0.8427829],"study_design_scores_gemma":[0.0001622181,0.001393181,0.01102518,0.1762563,0.006048126,0.001310715,0.0004836994,0.005628996,0.003976711,0.02543759,0.7679661,0.000311174],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0004229018,0.9949699,0.001757984,0.000630705,0.0002477858,0.00006504854,0.0002599325,0.00002469951,0.001621066],"genre_scores_gemma":[0.01061238,0.978402,0.008196708,0.0009027813,0.0003423341,0.0002677112,0.0006555005,0.00002726722,0.0005933968],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.9853157,"threshold_uncertainty_score":0.07765919,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1386965396708843,"score_gpt":0.4423723086560829,"score_spread":0.3036757689851987,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}