{"id":"W4410406004","doi":"10.1186/s12874-025-02561-x","title":"Comparison of methods for tuning machine learning model hyper-parameters: with application to predicting high-need high-cost health care users","year":2025,"lang":"en","type":"article","venue":"BMC Medical Research Methodology","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"Institute for Work & Health; Institute for Clinical Evaluative Sciences; University of Toronto","funders":"","keywords":"Machine learning; Bayesian optimization; Computer science; Artificial intelligence; Random forest; Gradient boosting; Overfitting; Boosting (machine learning); Extreme learning machine; Cross-validation; Calibration; Statistics; Mathematics; Artificial neural network","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","research_integrity"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.03824586,0.0002841619,0.001240266,0.001068167,0.0006589972,0.00006596764,0.00212781,0.0004695336,0.000009213492],"category_scores_gemma":[0.06924054,0.0002448483,0.0001043654,0.002007387,0.0003318152,0.0001362156,0.001094828,0.00291172,0.000003352225],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004454752,"about_ca_system_score_gemma":0.0033194,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006197397,"about_ca_topic_score_gemma":0.001652887,"domain_scores_codex":[0.9727648,0.02204322,0.00118123,0.001292734,0.001371661,0.001346341],"domain_scores_gemma":[0.9415341,0.05511523,0.0003982564,0.001087627,0.00106972,0.000795037],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005642145,0.00008870003,0.03870469,0.001797352,0.00005762615,0.000001051071,0.005780344,0.1559106,0.0007397057,0.03768137,0.0002257196,0.7584487],"study_design_scores_gemma":[0.001005044,0.001356018,0.001325361,0.0003655889,0.00001270449,0.00000496349,0.001325496,0.9893616,0.00124061,0.001831358,0.001993087,0.0001782111],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0168523,0.0008637549,0.9679337,0.01170798,0.0002423444,0.002095352,0.000008799369,0.0002210876,0.00007463997],"genre_scores_gemma":[0.2004635,0.00002942097,0.7978571,0.0007457325,0.00005613286,0.0007037431,0.00004449058,0.00003168781,0.00006819339],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.833451,"threshold_uncertainty_score":0.9993886,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3850406573113828,"score_gpt":0.6055089837650878,"score_spread":0.220468326453705,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}