{"id":"W4410766751","doi":"10.1371/journal.pone.0322048","title":"A new dataset for measuring the performance of blood vessel segmentation methods under distribution shifts","year":2025,"lang":"en","type":"article","venue":"PLoS ONE","topic":"Retinal Imaging and Analysis","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ottawa Hospital; University of Ottawa","funders":"Google Research; Fundação de Amparo à Pesquisa do Estado de São Paulo","keywords":"Computer science; Segmentation; Artificial intelligence; Pattern recognition (psychology); Metadata; Ground truth; Contrast (vision); Outlier; Annotation; Set (abstract data type); Artificial neural network; Task (project management); Convolutional neural network; Generalization; Transfer of learning; Sample (material); Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001180396,0.00124437,0.0007520513,0.002992618,0.000893192,0.001441955,0.001586581,0.001924753,0.002575541],"category_scores_gemma":[0.00433763,0.0003718203,0.001293947,0.0024144,0.0007430239,0.001379678,0.001325115,0.001743011,0.001859869],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001089114,"about_ca_system_score_gemma":0.001096902,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005276018,"about_ca_topic_score_gemma":0.01122534,"domain_scores_codex":[0.9982197,0.0002076415,0.0002147619,0.0005476914,0.0006432732,0.0001668923],"domain_scores_gemma":[0.9964632,0.0007779302,0.000362241,0.0009203344,0.001209018,0.0002673503],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002203105,0.002106178,0.04467092,0.003991772,0.0007090636,0.00124492,0.0006014211,0.05012165,0.1245304,0.007518979,0.4212873,0.3410142],"study_design_scores_gemma":[0.0006101833,0.001643403,0.2231831,0.0006581623,0.0004841252,0.006518194,0.001000096,0.1883208,0.132898,0.0120284,0.4321644,0.0004910733],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"dataset","genre_scores_codex":[0.4164204,0.005529312,0.1433447,0.001914289,0.00141059,0.001426664,0.3827321,0.02609953,0.02112235],"genre_scores_gemma":[0.2285415,0.001038574,0.1241146,0.0006160963,0.0002654575,0.00137733,0.6386009,0.001387134,0.004058273],"genre_candidate":"dataset","genre_consensus":null,"teacher_disagreement_score":0.005276018,"threshold_uncertainty_score":0.01049066,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06886594417720975,"score_gpt":0.3490845749832033,"score_spread":0.2802186308059936,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}