{"id":"W4408592579","doi":"10.1109/taslpro.2025.3552936","title":"A Multilingual Dataset (MultiMWP) and Benchmark for Math Word Problem Generation","year":2025,"lang":"en","type":"article","venue":"IEEE Transactions on Audio Speech and Language Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Benchmark (surveying); Natural language processing; Artificial intelligence; Word (group theory); Speech recognition; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001708756,0.003228143,0.0009619985,0.002837964,0.001336146,0.001593675,0.004285594,0.002636262,0.02021569],"category_scores_gemma":[0.009353977,0.0006179726,0.002227808,0.003677558,0.0007331255,0.002802299,0.002164874,0.00282595,0.01457121],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001821937,"about_ca_system_score_gemma":0.002413003,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01802337,"about_ca_topic_score_gemma":0.03807944,"domain_scores_codex":[0.9978989,0.0006052756,0.0002625931,0.000625466,0.0004542184,0.0001535858],"domain_scores_gemma":[0.9967322,0.001451437,0.0001733198,0.0007504612,0.0006352776,0.0002571522],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0007421049,0.001286634,0.004266465,0.003042371,0.0003080799,0.0005915219,0.0002656831,0.01961086,0.003589193,0.004202468,0.8366671,0.1254274],"study_design_scores_gemma":[0.002832528,0.001103546,0.0194017,0.0004947373,0.0002650907,0.00150214,0.001374172,0.2496225,0.02085528,0.01954811,0.6827451,0.0002550452],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.09695839,0.00290735,0.02929598,0.00254149,0.001129861,0.001804826,0.79285,0.03722206,0.03529],"genre_scores_gemma":[0.03422175,0.0002948378,0.04118479,0.0004181946,0.00009315355,0.001096513,0.9172283,0.0008408547,0.004621535],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.02021569,"threshold_uncertainty_score":0.0676282,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01368458297070991,"score_gpt":0.2964765663066384,"score_spread":0.2827919833359285,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}