{"id":"W4312261477","doi":"10.1109/cvpr52688.2022.00517","title":"Winoground: Probing Vision and Language Models for Visio-Linguistic Compositionality","year":2022,"lang":"en","type":"article","venue":"2022 IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","topic":"Multimodal Machine Learning Applications","field":"Computer Science","cited_by":185,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Principle of compositionality; Computer science; Set (abstract data type); Task (project management); Artificial intelligence; Natural language processing; Language model; Field (mathematics); State (computer science); Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003736268,0.004465636,0.001409772,0.004003144,0.002108563,0.004403759,0.00564254,0.004321734,0.01236508],"category_scores_gemma":[0.01498279,0.0008695145,0.002717929,0.002359469,0.002070962,0.006637028,0.00511719,0.004234019,0.01218563],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002813806,"about_ca_system_score_gemma":0.001833864,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02382137,"about_ca_topic_score_gemma":0.04241925,"domain_scores_codex":[0.9953696,0.001410412,0.0002843266,0.001551056,0.0009483427,0.0004363086],"domain_scores_gemma":[0.9932966,0.002680946,0.0003788262,0.002401344,0.0008720488,0.0003702777],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002056893,0.001285216,0.01468683,0.003215835,0.0006189026,0.0008452618,0.0007536715,0.02686524,0.01496031,0.01212634,0.7073715,0.2152139],"study_design_scores_gemma":[0.001337673,0.001344907,0.02203143,0.001043006,0.000421088,0.004068686,0.002689083,0.4710379,0.05289121,0.04023012,0.4024624,0.0004424668],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.2756926,0.01024918,0.1272124,0.005235222,0.002466207,0.002859758,0.4208225,0.09746896,0.0579933],"genre_scores_gemma":[0.2213479,0.0007626862,0.132909,0.002026225,0.0002904452,0.001300529,0.6286279,0.003756492,0.008978853],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02382137,"threshold_uncertainty_score":0.04736537,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04285904995147653,"score_gpt":0.3168681275049487,"score_spread":0.2740090775534722,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}