{"id":"W4402728032","doi":"10.1109/cvpr52733.2024.00386","title":"Emergent Open-Vocabulary Semantic Segmentation from Off-the-Shelf Vision-Language Models","year":2024,"lang":"en","type":"article","venue":"","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"Vector Institute; University of British Columbia","funders":"","keywords":"Computer science; Vocabulary; Segmentation; Natural language processing; Artificial intelligence; Image segmentation; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008893815,0.00236868,0.001280465,0.001263931,0.0004816155,0.001951309,0.00347818,0.002111699,0.005347282],"category_scores_gemma":[0.002895748,0.001044697,0.002040541,0.0009937155,0.001079186,0.004510208,0.00227079,0.002744787,0.004996684],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001565617,"about_ca_system_score_gemma":0.001595027,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01063449,"about_ca_topic_score_gemma":0.01891141,"domain_scores_codex":[0.9993258,0.00009801221,0.00002795058,0.00032702,0.0001214201,0.00009985662],"domain_scores_gemma":[0.9993373,0.0002434707,0.00005715045,0.0001587181,0.0001393371,0.00006399699],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006831946,0.000440351,0.002043475,0.0007580991,0.0003429154,0.0004823288,0.0005267244,0.2792227,0.06330973,0.0208061,0.03776513,0.5936193],"study_design_scores_gemma":[0.00003169615,0.00007084446,0.0002502511,0.00003294332,0.00003594351,0.00008752346,0.00006790245,0.9671488,0.009813645,0.01806499,0.004374817,0.00002063755],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04624232,0.001981387,0.9094489,0.0007915535,0.0003950806,0.0001895268,0.002019865,0.03209128,0.006840118],"genre_scores_gemma":[0.6098004,0.001157414,0.3563544,0.001619828,0.0002644669,0.0004139573,0.01250234,0.003271312,0.014616],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01063449,"threshold_uncertainty_score":0.02114516,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01812614014927617,"score_gpt":0.3248814039916032,"score_spread":0.306755263842327,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}