{"id":"W4404781890","doi":"10.18653/v1/2024.findings-emnlp.191","title":"Are Large Vision Language Models up to the Challenge of Chart Comprehension and Reasoning","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada; Centre International de Recherche sur le Cancer","keywords":"Computer science; Comprehension; Chart; Natural language processing; Artificial intelligence; Programming language; Statistics; Mathematics","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003053909,0.00009175057,0.0001139349,0.00008699068,0.00007014003,0.0001395116,0.000413585,0.00004492645,0.000008894044],"category_scores_gemma":[0.00002545886,0.00005319017,0.00002796189,0.0002497371,0.00001455448,0.0003870488,0.0005369627,0.0001345834,0.000007833428],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001150049,"about_ca_system_score_gemma":0.00001002617,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002626519,"about_ca_topic_score_gemma":0.00002706118,"domain_scores_codex":[0.9992145,0.00003269347,0.0001175698,0.0002875685,0.0001980523,0.0001496414],"domain_scores_gemma":[0.9994484,0.0000577614,0.00003934386,0.0003633216,0.00004780192,0.00004341672],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0000107867,0.0000376298,0.00001737362,0.0002985945,0.00001783789,0.00008054036,0.02451873,0.00002578964,0.009344839,0.7935653,0.009096955,0.1629856],"study_design_scores_gemma":[0.0001568086,0.00014159,0.0001090108,0.001399264,0.000009802354,0.00003770912,0.000805108,0.9282088,0.0230953,0.04075194,0.004983269,0.0003014433],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01642934,0.02939633,0.9438659,0.007870797,0.0002064516,0.0002237579,0.000005140513,0.0008992743,0.001103021],"genre_scores_gemma":[0.9175173,0.00005269588,0.08169919,0.0004124605,0.00003196494,0.000005888157,8.081212e-7,0.000007897546,0.0002717935],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.928183,"threshold_uncertainty_score":0.2169032,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01876255332833101,"score_gpt":0.3026275532543308,"score_spread":0.2838649999259998,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}