{"id":"W7082278849","doi":"10.48448/dzw3-dg27","title":"ChartQAPro: A More Diverse and Challenging Benchmark for Chart Question Answering","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Military Technology and Strategies","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; York University","funders":"","keywords":"Chart; Question answering; Interpretability; Pie chart; Complement (music); Benchmark (surveying)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007493858,0.002995746,0.0010845,0.004553813,0.001777823,0.004568442,0.004033558,0.003630327,0.01604045],"category_scores_gemma":[0.05011183,0.0005697577,0.001961587,0.004067371,0.001361539,0.008734684,0.003789063,0.003736682,0.007670744],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003314221,"about_ca_system_score_gemma":0.005537853,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.04473481,"about_ca_topic_score_gemma":0.04353991,"domain_scores_codex":[0.9895746,0.003920726,0.0009207163,0.002366581,0.002697867,0.0005195233],"domain_scores_gemma":[0.9660279,0.01848584,0.001030121,0.005565527,0.007538424,0.001352279],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001316897,0.001649121,0.01358071,0.003929372,0.0003418251,0.000633261,0.001603499,0.05868537,0.00950359,0.01907961,0.5797563,0.3099204],"study_design_scores_gemma":[0.0007135834,0.000974254,0.01107642,0.0006572008,0.0001502685,0.0006420698,0.001908179,0.5483079,0.02698484,0.04042641,0.3678674,0.0002915133],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"software","genre_gemma":"methods","genre_scores_codex":[0.2021956,0.008532233,0.2467629,0.009420847,0.003894634,0.003130269,0.1664922,0.2882846,0.07128672],"genre_scores_gemma":[0.3107481,0.00134649,0.3145418,0.003268785,0.0003551986,0.00125334,0.3477116,0.007827491,0.01294718],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.04473481,"threshold_uncertainty_score":0.08894885,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01036467753816499,"score_gpt":0.2586344229976558,"score_spread":0.2482697454594908,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}