{"id":"W4409347925","doi":"10.1609/aaai.v39i23.34673","title":"Evaluating LLM Reasoning in the Operations Research Domain with ORQA","year":2025,"lang":"en","type":"article","venue":"Proceedings of the AAAI Conference on Artificial Intelligence","topic":"Business Process Modeling and Analysis","field":"Business, Management and Accounting","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia; Huawei Technologies (Canada)","funders":"","keywords":"Domain (mathematical analysis); Computer science; Management science; Mathematics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003287405,0.0001685814,0.0002150139,0.0004542712,0.0006912648,0.0008235875,0.001180588,0.00006377861,0.00005117654],"category_scores_gemma":[0.00122028,0.0000963088,0.00006139612,0.003392154,0.0003006244,0.0005581618,0.000225571,0.0005045001,0.0000386274],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003423947,"about_ca_system_score_gemma":0.0001264815,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001044937,"about_ca_topic_score_gemma":0.0005331766,"domain_scores_codex":[0.9980636,0.00002349635,0.000449251,0.0003902,0.0007315434,0.000341944],"domain_scores_gemma":[0.9976822,0.0001101633,0.0001400258,0.0002369385,0.001822921,0.000007824667],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001069748,0.0001863059,0.001466328,0.000132915,0.00001730796,4.851595e-7,0.0007586472,0.00643422,0.004275415,0.9663823,0.00009688736,0.02014222],"study_design_scores_gemma":[0.00009827858,0.00006015399,0.0007586869,0.001983549,0.0000812068,0.00000120336,0.01845176,0.6688399,0.008229026,0.300983,0.0002313667,0.0002818456],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9448921,0.00004487062,0.002856186,0.01543397,0.00009821164,0.0004944861,7.719161e-7,0.00004219719,0.03613717],"genre_scores_gemma":[0.9981286,0.00001178938,0.0007439667,0.0006297846,0.0001465964,0.00009302299,0.000001255716,0.00001097496,0.0002339991],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6653993,"threshold_uncertainty_score":0.7941873,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2101254107643576,"score_gpt":0.40640091708382,"score_spread":0.1962755063194624,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}