{"id":"W4416159353","doi":"10.48550/arxiv.2511.07396","title":"C3PO: Optimized Large Language Model Cascades with Probabilistic Cost Constraints for Reasoning","year":2025,"lang":"","type":"preprint","venue":"ArXiv.org","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Probabilistic logic; Regret; Inference; Generalization; Scalability; Cascade; Set (abstract data type)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002665958,0.002200462,0.001398411,0.0008637864,0.000880626,0.001436232,0.003973087,0.00216053,0.006604954],"category_scores_gemma":[0.0122357,0.00115096,0.001667794,0.0008310407,0.001075534,0.003609237,0.002702135,0.004195089,0.0022591],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002399724,"about_ca_system_score_gemma":0.003103923,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01219517,"about_ca_topic_score_gemma":0.02674146,"domain_scores_codex":[0.9983019,0.0004175285,0.00006977795,0.0005044442,0.0005516557,0.0001547655],"domain_scores_gemma":[0.9966628,0.001980986,0.0001923438,0.0006181311,0.000384684,0.0001610396],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003011951,0.0003362505,0.001467498,0.000315807,0.000181416,0.0002163789,0.000131233,0.7838256,0.006034227,0.02547558,0.02820934,0.1535054],"study_design_scores_gemma":[0.00001299478,0.00001616767,0.0000488981,0.000004601362,0.000007841743,0.00001187057,0.000004269332,0.9916284,0.0007475532,0.006963469,0.000549283,0.000004623968],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02250551,0.0007048619,0.9573483,0.0008450345,0.0001290968,0.000276124,0.0007271118,0.01264245,0.004821396],"genre_scores_gemma":[0.4088409,0.0004793408,0.5751967,0.001132817,0.0002322503,0.0006280838,0.003629753,0.002258732,0.007601311],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01219517,"threshold_uncertainty_score":0.02424836,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04827555444929198,"score_gpt":0.3026207778427652,"score_spread":0.2543452233934732,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}