{"id":"W4285255856","doi":"10.18653/v1/2022.findings-acl.177","title":"ChartQA: A Benchmark for Question Answering about Charts with Visual and Logical Reasoning","year":2022,"lang":"en","type":"article","venue":"Findings of the Association for Computational Linguistics: ACL 2022","topic":"Data Visualization and Analytics","field":"Computer Science","cited_by":246,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Question answering; Benchmark (surveying); Vocabulary; Chart; Visual reasoning; Artificial intelligence; Qualitative reasoning; Variety (cybernetics); Natural language processing; Linguistics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004111089,0.003212562,0.001137991,0.006394983,0.00126995,0.003200818,0.00385166,0.003255888,0.01390989],"category_scores_gemma":[0.04093298,0.0005008758,0.002468993,0.005371743,0.001048974,0.006262931,0.003256219,0.00258332,0.006823266],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00278679,"about_ca_system_score_gemma":0.003654334,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0253598,"about_ca_topic_score_gemma":0.02904454,"domain_scores_codex":[0.9921536,0.002666268,0.001132578,0.001637871,0.002019465,0.0003902448],"domain_scores_gemma":[0.9761578,0.01578541,0.001055666,0.002476447,0.003621586,0.0009031366],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00115254,0.0012425,0.01147461,0.009581684,0.0003436166,0.0006114886,0.00168374,0.03366618,0.008763603,0.01714311,0.6539646,0.2603723],"study_design_scores_gemma":[0.0008232821,0.0008586244,0.01740245,0.001327336,0.0002316712,0.001029331,0.002954575,0.3833861,0.02238224,0.06013995,0.5092036,0.0002607812],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.1032715,0.01339643,0.1933505,0.006407084,0.00144971,0.004203472,0.4923486,0.1456257,0.03994691],"genre_scores_gemma":[0.1403304,0.002152541,0.266473,0.001352935,0.0002584041,0.00224625,0.5780025,0.002694445,0.00648953],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0253598,"threshold_uncertainty_score":0.0504244,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.00984609113906177,"score_gpt":0.2772540827683285,"score_spread":0.2674079916292668,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}