{"id":"W4410766751","doi":"10.1371/journal.pone.0322048","title":"A new dataset for measuring the performance of blood vessel segmentation methods under distribution shifts","year":2025,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Retinal Imaging and Analysis","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ottawa Hospital; University of Ottawa","funders":"Google Research; Fundação de Amparo à Pesquisa do Estado de São Paulo","keywords":"Computer science; Segmentation; Artificial intelligence; Pattern recognition (psychology); Metadata; Ground truth; Contrast (vision); Outlier; Annotation; Set (abstract data type); Artificial neural network; Task (project management); Convolutional neural network; Generalization; Transfer of learning; Sample (material); Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003339845,0.0000515869,0.0001503659,0.00003410975,0.0000590318,0.00001054953,0.0000553229,0.00001756629,0.000008626115],"category_scores_gemma":[0.0001133754,0.00003625203,0.00003738846,0.000191285,0.00001970731,0.00004804197,0.00001711214,0.00006068108,0.000001164345],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001782866,"about_ca_system_score_gemma":0.00004910901,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002900491,"about_ca_topic_score_gemma":0.000001014723,"domain_scores_codex":[0.999501,0.00003626784,0.0001445789,0.00009978098,0.000137852,0.00008047589],"domain_scores_gemma":[0.9995726,0.00009824432,0.00005768013,0.0001767511,0.00007012852,0.000024562],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002925269,0.001086791,0.01503853,0.001390364,0.002581943,6.362056e-7,0.0001836735,0.00004992969,0.9504387,0.0004711454,0.00474963,0.02371609],"study_design_scores_gemma":[0.0009598996,0.0001026079,0.01060571,0.0005799041,0.004327287,7.348876e-7,0.0001171025,0.009392442,0.9733911,0.0002244529,0.0002454291,0.00005328995],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7676055,0.0006098164,0.2248291,0.006107781,0.0000228286,0.0003809803,0.0002917019,0.00002276528,0.0001295287],"genre_scores_gemma":[0.9446693,0.00008581285,0.0524577,0.0002518555,0.00005493687,0.00002580407,0.001995668,0.000005787243,0.0004531016],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1770638,"threshold_uncertainty_score":0.1478315,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06886594417720975,"score_gpt":0.3490845749832033,"score_spread":0.2802186308059936,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}