{"id":"W1994343730","doi":"10.1002/sim.1723","title":"Three validation metrics for automated probabilistic image segmentation of brain tumours","year":2004,"lang":"en","type":"article","venue":"Statistics in Medicine","topic":"Medical Image Segmentation Techniques","field":"Computer Science","cited_by":95,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"National Center for Research Resources; U.S. National Library of Medicine; National Cancer Institute; Agency for Healthcare Research and Quality; National Institutes of Health","keywords":"Segmentation; Computer science; Artificial intelligence; Markov random field; Metric (unit); Similarity (geometry); Pixel; Pattern recognition (psychology); Gold standard (test); Image segmentation; Image (mathematics); Mathematics; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04576191,0.00138324,0.001526687,0.00582204,0.001159661,0.001908723,0.002048536,0.002850899,0.0003840539],"category_scores_gemma":[0.1691265,0.0006635835,0.001247175,0.002178127,0.002465686,0.001812485,0.002453495,0.001157233,0.0001628791],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002447379,"about_ca_system_score_gemma":0.001657439,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002962226,"about_ca_topic_score_gemma":0.002991913,"domain_scores_codex":[0.9755824,0.0136662,0.002295922,0.001790143,0.006054156,0.0006111562],"domain_scores_gemma":[0.8091411,0.1495309,0.01538234,0.008478515,0.01653411,0.0009329944],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002271682,0.0003102642,0.07433685,0.0007108554,0.0008144088,0.0002842391,0.001033476,0.6197768,0.02464985,0.01336468,0.001472613,0.2609742],"study_design_scores_gemma":[0.00005138632,0.0003887696,0.01970055,0.00008075312,0.00005670908,0.0003283728,0.00007370515,0.9557498,0.01882156,0.004112745,0.0005164807,0.0001192193],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3595977,0.001580956,0.6354725,0.0002067358,0.00004460042,0.0003374175,0.0002922465,0.001349465,0.001118424],"genre_scores_gemma":[0.7264544,0.0001592264,0.2720129,0.00005652184,0.00002095587,0.0003611013,0.0005830806,0.0001428174,0.0002089459],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.04576191,"threshold_uncertainty_score":0.242015,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02776990431574732,"score_gpt":0.3596305153886217,"score_spread":0.3318606110728743,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}