{"id":"W4377002285","doi":"10.59275/j.melba.2022-db5c","title":"Label fusion and training methods for reliable representation of inter-rater uncertainty","year":2023,"lang":"en","type":"article","venue":"The Journal of Machine Learning for Biomedical Imaging","topic":"Medical Imaging and Analysis","field":"Engineering","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal; Mila - Quebec Artificial Intelligence Institute","funders":"Fonds de recherche du Québec – Nature et technologies; Canadian Institutes of Health Research; Natural Sciences and Engineering Research Council of Canada; Institut de Valorisation des Données; Craig H. Neilsen Foundation; Canada First Research Excellence Fund; Nvidia","keywords":"Segmentation; Artificial intelligence; Computer science; Ground truth; Machine learning; Task (project management); Pattern recognition (psychology)","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03616056,0.002636533,0.002659104,0.003583015,0.001698452,0.003415501,0.00341544,0.004108119,0.002885934],"category_scores_gemma":[0.07527954,0.0009044221,0.002406852,0.002349879,0.002884374,0.004368023,0.005391567,0.006383794,0.002119108],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002330921,"about_ca_system_score_gemma":0.002861915,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003502611,"about_ca_topic_score_gemma":0.003836736,"domain_scores_codex":[0.9770899,0.009567721,0.00162227,0.006243732,0.004688437,0.0007878682],"domain_scores_gemma":[0.9551587,0.02384221,0.005719242,0.00806762,0.006546684,0.0006655768],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001331355,0.0003847301,0.01232362,0.0007659299,0.0008812239,0.0003112263,0.00243342,0.1938944,0.01645514,0.01441867,0.01199382,0.7448064],"study_design_scores_gemma":[0.00009530254,0.0003732448,0.005791005,0.0002625411,0.0002001602,0.0003233802,0.0002422093,0.9310588,0.01997221,0.03483023,0.00669425,0.0001566869],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02054622,0.001760517,0.9718748,0.0006387239,0.0001887598,0.0002304473,0.0003486519,0.003174005,0.00123793],"genre_scores_gemma":[0.4363136,0.0009265417,0.553354,0.0009544075,0.0005686476,0.001073017,0.002363544,0.001501301,0.002945029],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03616056,"threshold_uncertainty_score":0.1912376,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03216285539838912,"score_gpt":0.365834025024052,"score_spread":0.3336711696256628,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}