{"id":"W4410733269","doi":"10.1016/j.compbiomed.2025.110375","title":"Beyond Accuracy: Evaluating certainty of AI models for brain tumour detection","year":2025,"lang":"en","type":"article","venue":"Computers in Biology and Medicine","topic":"Brain Tumor Detection and Classification","field":"Neuroscience","cited_by":3,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université de Moncton","funders":"King Saud University","keywords":"Certainty; Computer science; Artificial intelligence; Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006056251,0.0000718544,0.0001695703,0.0001860776,0.0000837256,0.000002970187,0.00008886506,0.0000961831,0.000003337794],"category_scores_gemma":[0.001586374,0.00005893073,0.00001949075,0.0002305078,0.0002265474,0.00003834422,0.00002957905,0.0001661194,2.072598e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002469195,"about_ca_system_score_gemma":0.00002860297,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001513362,"about_ca_topic_score_gemma":0.00001451945,"domain_scores_codex":[0.9991714,0.0001758461,0.0002421939,0.0002547094,0.00004164649,0.0001141986],"domain_scores_gemma":[0.9982876,0.001460358,0.00009013925,0.00009807675,0.00004030335,0.00002354667],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001216407,0.00001771433,0.0001395694,0.0000463872,0.000003357824,3.290765e-7,0.0002416327,0.0001703343,0.7616939,0.06794792,0.0004264127,0.1691908],"study_design_scores_gemma":[0.002456715,0.0006064851,0.003190903,0.0001361817,0.0000142434,0.00001346174,0.0001506094,0.5528306,0.1447005,0.2950265,0.0007782454,0.00009551193],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3668664,0.0001356795,0.6098477,0.02032779,0.00130104,0.000486612,0.000003596533,0.0000433378,0.0009878426],"genre_scores_gemma":[0.9922974,0.00002259715,0.000500488,0.007040808,0.00005403759,0.00003048644,0.000003058121,0.000003003539,0.00004818114],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6254309,"threshold_uncertainty_score":0.2403125,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06222629019690736,"score_gpt":0.3918231036985212,"score_spread":0.3295968135016139,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}