{"id":"W4400748060","doi":"10.1016/j.modpat.2024.100563","title":"Evaluation of Artificial Intelligence-Based Gleason Grading Algorithms “in the Wild”","year":2024,"lang":"en","type":"article","venue":"Modern Pathology","topic":"AI in cancer detection","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"Princess Margaret Cancer Centre","funders":"Radboud Universitair Medisch Centrum; Health~Holland","keywords":"Grading (engineering); Computer science; Pathology; Algorithm; Artificial intelligence; Medical physics; Medicine; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003208706,0.00008301654,0.0001022804,0.0001825276,0.00004373094,0.00005745383,0.0004363899,0.0000743261,0.00000903723],"category_scores_gemma":[0.00007869558,0.00006577407,0.00004937863,0.0004824208,0.0000642611,0.0001570958,0.0000413505,0.0001706886,0.00001773548],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001436333,"about_ca_system_score_gemma":0.0001630063,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002883521,"about_ca_topic_score_gemma":0.00005925982,"domain_scores_codex":[0.9982546,0.0004943769,0.0002297821,0.0003429821,0.0005099908,0.0001682889],"domain_scores_gemma":[0.999317,0.0001612744,0.00004889014,0.0003669809,0.00009203069,0.00001378398],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000003112076,0.00002592053,0.00002965325,0.00001012339,0.00000347326,0.00002366289,0.002344865,0.0203515,0.001687672,0.007788135,0.00001926117,0.9677126],"study_design_scores_gemma":[0.00003100462,0.00007891283,0.0002338926,0.00002260593,0.00001474794,0.00002988026,0.00003958809,0.827286,0.004748403,0.1674189,0.00004258552,0.00005346642],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.05543621,0.0008564283,0.9411983,0.001108301,0.0008955281,0.0002024795,0.000001444499,0.00006734899,0.0002339412],"genre_scores_gemma":[0.9905251,0.000007145608,0.00919638,0.000103207,0.0000915045,0.00006580483,0.000001312893,0.000006668565,0.000002854256],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9676592,"threshold_uncertainty_score":0.2682189,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09863510812397899,"score_gpt":0.341038808473111,"score_spread":0.242403700349132,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}