{"id":"W7105692598","doi":"10.1109/ijcnn64981.2025.11227916","title":"Model Cascading for Code: A Cascaded Black-Box Multi-Model Framework for Cost-Efficient Code Completion with Self-Testing","year":2025,"lang":"","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"York University","keywords":"Code (set theory); Heuristic; Server; Sensitivity (control systems); Test case; Source code; Computational complexity theory; Component (thermodynamics)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004580134,0.0021675,0.001652977,0.001433312,0.0009029739,0.00184081,0.005479618,0.002098745,0.006088172],"category_scores_gemma":[0.01934634,0.001468145,0.002389164,0.000759605,0.002237285,0.003303532,0.003683235,0.003562672,0.001322869],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001778701,"about_ca_system_score_gemma":0.002912717,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006460168,"about_ca_topic_score_gemma":0.007750281,"domain_scores_codex":[0.9967056,0.001279243,0.0001456129,0.0006057677,0.0008860875,0.0003776195],"domain_scores_gemma":[0.9882676,0.007286732,0.0007494765,0.002109168,0.001073165,0.0005138947],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002127363,0.00020753,0.002487442,0.000213306,0.00009007101,0.0002963336,0.0002186882,0.8763492,0.00490596,0.01820756,0.002534901,0.09427623],"study_design_scores_gemma":[0.0000123057,0.0000265154,0.00003614209,0.000005913819,0.000007344616,0.00001830403,0.000007214555,0.9926011,0.0007816168,0.006229032,0.0002692614,0.000005180451],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01669367,0.0002169174,0.9750574,0.0002475082,0.00003775701,0.0001764192,0.00007637365,0.006445697,0.001048238],"genre_scores_gemma":[0.425871,0.0001446296,0.5692305,0.0003034979,0.00004476129,0.0004638206,0.0004039405,0.001641433,0.001896384],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006460168,"threshold_uncertainty_score":0.02422237,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1273860547350612,"score_gpt":0.3602348113149539,"score_spread":0.2328487565798927,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}