{"id":"W7082278849","doi":"10.48448/dzw3-dg27","title":"ChartQAPro: A More Diverse and Challenging Benchmark for Chart Question Answering","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Military Technology and Strategies","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; York University","funders":"","keywords":"Chart; Question answering; Interpretability; Pie chart; Complement (music); Benchmark (surveying)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002328919,0.000214501,0.0002100144,0.0005546449,0.0001697736,0.00003235527,0.0002591385,0.0002310469,0.00009548841],"category_scores_gemma":[0.00003675962,0.0002165633,0.00002743253,0.0002230726,0.0004057222,0.0001355524,0.0000714924,0.0002009552,0.000006086569],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004747864,"about_ca_system_score_gemma":0.00006199459,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005108576,"about_ca_topic_score_gemma":0.0001027476,"domain_scores_codex":[0.9990556,0.000004585635,0.0001358526,0.0003560733,0.0001270075,0.0003208483],"domain_scores_gemma":[0.999625,0.00002504945,0.00002840297,0.0002382078,0.00003068272,0.00005260502],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002600589,0.00009666783,0.0004781992,0.004134658,0.0003136195,0.00004315162,0.001727668,0.004941743,0.00384527,0.6866506,0.06399024,0.2337521],"study_design_scores_gemma":[0.001525014,0.0003227863,0.0008372658,0.002933616,0.0001992053,0.00003433875,0.00609988,0.6728012,0.001031997,0.03275913,0.2790075,0.00244807],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.001969775,0.01468749,0.07588089,0.0009815946,0.002557589,0.00165466,0.0002815396,0.003080064,0.8989064],"genre_scores_gemma":[0.7839507,0.01131782,0.0418593,0.0002198649,0.001101957,0.0004085681,0.0002748791,0.0005712392,0.1602957],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.7819809,"threshold_uncertainty_score":0.8831195,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01036467753816499,"score_gpt":0.2586344229976558,"score_spread":0.2482697454594908,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}