{"id":"W4415655710","doi":"10.1007/978-3-032-08970-0_8","title":"Simulating Inter-observer Variability Across Clinical Experience Levels for Brain Tumour Segmentation","year":2025,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Hotchkiss Brain Institute; Alberta Children's Hospital; University of Calgary","funders":"","keywords":"Annotation; Segmentation; Trustworthiness; Deep learning; Image segmentation; Observer (physics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00242377,0.0004518746,0.0004556324,0.0004373707,0.0001835923,0.0007631857,0.0007602901,0.001119912,0.001184714],"category_scores_gemma":[0.01384224,0.0003727256,0.0008460751,0.0003947997,0.0004124807,0.0004503152,0.0006546839,0.000693356,0.000308361],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000655116,"about_ca_system_score_gemma":0.0005011131,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007410867,"about_ca_topic_score_gemma":0.006298972,"domain_scores_codex":[0.9990313,0.0004348393,0.00004729415,0.0002319554,0.0001480064,0.0001065621],"domain_scores_gemma":[0.9818999,0.01584485,0.0004401677,0.0006922255,0.0009231403,0.0001997223],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002863385,0.0005455577,0.02736772,0.0001940408,0.0003722659,0.0003025077,0.001211884,0.8309252,0.02478173,0.001263483,0.001980879,0.1081914],"study_design_scores_gemma":[0.0000278823,0.0002479146,0.0114366,0.00001012932,0.00005305609,0.00009064996,0.00007696687,0.9827574,0.004405465,0.0006859586,0.0001823256,0.00002554426],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8454134,0.0002518115,0.1518517,0.0001380374,0.00007884193,0.00008192198,0.0004378961,0.0005913012,0.001155199],"genre_scores_gemma":[0.9792576,0.00004552815,0.01903117,0.00003632902,0.00001232339,0.00005473902,0.0004371984,0.0000931931,0.001031867],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007410867,"threshold_uncertainty_score":0.01473546,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.235101572048581,"score_gpt":0.502204069754961,"score_spread":0.2671024977063801,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}