{"id":"W4410733269","doi":"10.1016/j.compbiomed.2025.110375","title":"Beyond Accuracy: Evaluating certainty of AI models for brain tumour detection","year":2025,"lang":"en","type":"article","venue":"Computers in Biology and Medicine","topic":"Brain Tumor Detection and Classification","field":"Neuroscience","cited_by":3,"is_retracted":false,"has_abstract":false,"ca_institutions":"Université de Moncton","funders":"King Saud University","keywords":"Certainty; Computer science; Artificial intelligence; Machine learning; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03759907,0.002368152,0.002760928,0.004980412,0.001134255,0.006475323,0.003327368,0.005255059,0.002198632],"category_scores_gemma":[0.2183345,0.0009074952,0.002240082,0.001794772,0.002931577,0.007766655,0.003584267,0.004476205,0.0005859481],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002782203,"about_ca_system_score_gemma":0.001852346,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007436745,"about_ca_topic_score_gemma":0.004655163,"domain_scores_codex":[0.9798167,0.00869614,0.001814716,0.003254394,0.005496365,0.0009215833],"domain_scores_gemma":[0.7255886,0.2427469,0.01032414,0.009524192,0.009302955,0.002513253],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"not_applicable","study_design_scores_codex":[0.003785511,0.0003271267,0.114102,0.0008429359,0.00182757,0.0003648623,0.0005854849,0.7059136,0.001872748,0.01643917,0.004604065,0.149335],"study_design_scores_gemma":[0.00008119238,0.0005369707,0.006229726,0.0001334818,0.0002988167,0.0003392897,0.00011867,0.9598294,0.001841221,0.02965049,0.000873569,0.00006713481],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.4350951,0.0139492,0.5296263,0.007279336,0.0006168497,0.0003182902,0.002408741,0.002180697,0.008525471],"genre_scores_gemma":[0.9758326,0.0006700107,0.02118422,0.0004050673,0.0003372334,0.00003469834,0.0007727362,0.0001043648,0.0006591069],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.03759907,"threshold_uncertainty_score":0.1988453,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06222629019690736,"score_gpt":0.3918231036985212,"score_spread":0.3295968135016139,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}