{"id":"W7116141524","doi":"10.1007/s10489-025-07044-6","title":"CMCTS: A Constrained Monte Carlo Tree Search framework for mathematical reasoning in large language model","year":2025,"lang":"en","type":"article","venue":"Applied Intelligence","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"","keywords":"Monte Carlo tree search; Tree (set theory); Action (physics); Action selection; Monte Carlo method; Baseline (sea); Selection (genetic algorithm)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003837985,0.001066476,0.001892421,0.00202991,0.001100182,0.002942674,0.004399037,0.002544409,0.01397947],"category_scores_gemma":[0.02290979,0.001130137,0.002148692,0.002660032,0.001649784,0.00367134,0.003252818,0.003512361,0.002542463],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001819144,"about_ca_system_score_gemma":0.00411533,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01580671,"about_ca_topic_score_gemma":0.02337698,"domain_scores_codex":[0.9976996,0.001203488,0.0001166317,0.0002561802,0.0006010577,0.0001230524],"domain_scores_gemma":[0.9904246,0.00767364,0.0002630559,0.0006442606,0.0006737101,0.0003207158],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002251677,0.0001331811,0.0007024196,0.0002331175,0.0001830195,0.0001565774,0.0002112033,0.6136288,0.0007482368,0.2683907,0.01150225,0.1038853],"study_design_scores_gemma":[0.00001728764,0.000006565816,0.00001786562,0.000008821969,0.000008602704,0.000009181377,0.000007272043,0.9324458,0.0001150048,0.06613462,0.001223829,0.000005116224],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.001142199,0.0001088155,0.9963935,0.0001515054,0.0000272937,0.00005098873,0.0001849895,0.001002307,0.000938512],"genre_scores_gemma":[0.1303322,0.000363804,0.8624943,0.0003739065,0.0001761116,0.0005982596,0.001324393,0.001127034,0.003209944],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01580671,"threshold_uncertainty_score":0.04676598,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02824518937086059,"score_gpt":0.3237130342735292,"score_spread":0.2954678449026686,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}