{"id":"W4393148525","doi":"10.1609/aaai.v38i5.28272","title":"SeTformer Is What You Need for Vision and Language","year":2024,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Speech and dialogue systems","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"École de Technologie Supérieure","funders":"Vetenskapsrådet; Knut och Alice Wallenbergs Stiftelse","keywords":"Computer science; Linguistics; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0004448233,0.0001700872,0.0001892133,0.0001259849,0.0001207189,0.001185203,0.0008344788,0.00008066691,0.00002114847],"category_scores_gemma":[0.0001401382,0.0001124739,0.0001042856,0.0004142244,0.0001252767,0.001077496,0.0002111545,0.0001370604,0.00006983388],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002227228,"about_ca_system_score_gemma":0.00007176205,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003861525,"about_ca_topic_score_gemma":0.000004489741,"domain_scores_codex":[0.9986778,0.000006217472,0.000324043,0.0004423082,0.0002854298,0.0002642337],"domain_scores_gemma":[0.9992685,0.0001068955,0.000105093,0.0002005971,0.0002396188,0.0000792893],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00004546633,0.00003664583,0.00002191571,0.0001784546,0.00001697177,6.422043e-7,0.006941416,0.000001503572,0.06568083,0.5911453,0.001599346,0.3343315],"study_design_scores_gemma":[0.00003489593,0.0003304127,0.00003670192,0.0008704659,0.00001550802,0.000011689,0.003872643,0.1005614,0.7420316,0.1501655,0.001820504,0.00024856],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7242061,0.004416541,0.1807152,0.05385433,0.009138077,0.003972596,0.00007506508,0.0009359868,0.02268609],"genre_scores_gemma":[0.9966881,0.0001181398,0.002088606,0.0003498179,0.0001193654,0.00003253426,7.140322e-7,0.00001120368,0.0005914896],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6763508,"threshold_uncertainty_score":0.9998516,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05020884113658372,"score_gpt":0.3140827375632839,"score_spread":0.2638738964267002,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}