{"id":"W4387358408","doi":"10.1101/2023.10.03.560782","title":"The Bias of Using Cross-Validation in Genomic Predictions and Its Correction","year":2023,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Genetic and phenotypic traits in livestock","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Alberta Children's Hospital; Churchill Northern Studies Centre; University of Calgary","funders":"","keywords":"Elastic net regularization; Regularization (linguistics); Cross-validation; Computer science; Ridge; Lasso (programming language); Sample size determination; Statistics; Generalization; Mean squared error; Mathematics; Feature selection; Algorithm; Artificial intelligence; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04528522,0.001962922,0.001371701,0.001769084,0.001259174,0.001714944,0.002058883,0.001880928,0.001071949],"category_scores_gemma":[0.1062177,0.0005692838,0.001199668,0.001526919,0.001868853,0.001946414,0.002365891,0.003063973,0.0005367909],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001341835,"about_ca_system_score_gemma":0.002139149,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007052704,"about_ca_topic_score_gemma":0.006483921,"domain_scores_codex":[0.9755498,0.01333722,0.001870713,0.004664693,0.003894584,0.0006830835],"domain_scores_gemma":[0.9025121,0.06569637,0.005251236,0.01260619,0.01321544,0.0007186451],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001439062,0.0003867063,0.1257504,0.0009857991,0.002591103,0.0008228417,0.0008377235,0.5270941,0.02733492,0.0191264,0.01515572,0.2784753],"study_design_scores_gemma":[0.0000516506,0.0002000432,0.01728657,0.0003666163,0.0001830209,0.0003377335,0.0001320806,0.9435182,0.02144991,0.01170878,0.004654683,0.0001107412],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2263447,0.005207293,0.7577135,0.001688274,0.001430459,0.0001596354,0.0008034133,0.003681157,0.002971564],"genre_scores_gemma":[0.8801687,0.0003451844,0.1144803,0.0009971338,0.0001462583,0.0001625632,0.001219213,0.0009527539,0.001527881],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.04528522,"threshold_uncertainty_score":0.239494,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03182024183289273,"score_gpt":0.2603240198815394,"score_spread":0.2285037780486466,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}