{"id":"W4417458211","doi":"10.1038/s41746-025-02237-2","title":"SLEEPYLAND: trust begins with fair evaluation of automatic sleep staging models","year":2025,"lang":"en","type":"article","venue":"npj Digital Medicine","topic":"EEG and Brain-Computer Interfaces","field":"Neuroscience","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Center for Research Resources; National Institute of Biomedical Imaging and Bioengineering; National Institute of Arthritis and Musculoskeletal and Skin Diseases; National Institute of Nursing Research; National Heart, Lung, and Blood Institute; National Institute on Aging; University of California, Davis; National Center for Advancing Translational Sciences; Johns Hopkins Bloomberg School of Public Health; Danmarks Tekniske Universitet; York University; Jazz Pharmaceuticals; National Institutes of Health; H. Lundbeck A/S; University of Minnesota; Lundbeckfonden; Case Western Reserve University; Johns Hopkins University; University of Washington; Eunice Kennedy Shriver National Institute of Child Health and Human Development; American Sleep Medicine Foundation","keywords":"Ambiguity; Ensemble forecasting; Ensemble learning; Deep learning; Multiple Models; Sleep (system call)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01068733,0.003079758,0.001410138,0.001390872,0.00106396,0.002874836,0.003188292,0.002383633,0.00746615],"category_scores_gemma":[0.03964668,0.0009487583,0.001450643,0.000595434,0.00128008,0.003504788,0.004602205,0.002798004,0.004931352],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001647774,"about_ca_system_score_gemma":0.00312952,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01298365,"about_ca_topic_score_gemma":0.02633172,"domain_scores_codex":[0.9954703,0.001619696,0.0003348558,0.001243101,0.0009248684,0.00040717],"domain_scores_gemma":[0.9917903,0.003513758,0.0004246428,0.002287631,0.001544139,0.000439544],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.003315729,0.0007298651,0.04708789,0.001485819,0.001901691,0.0006830753,0.0006789486,0.1625753,0.01741442,0.007433121,0.2101748,0.5465192],"study_design_scores_gemma":[0.0002565572,0.000688786,0.006630665,0.0003309588,0.0001775512,0.0003763864,0.0001886253,0.9378298,0.01543504,0.01470619,0.02326874,0.0001107747],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2072564,0.006351409,0.6430778,0.005184803,0.002680318,0.001186409,0.01196602,0.1024825,0.01981435],"genre_scores_gemma":[0.8160072,0.0008903614,0.1428547,0.001802745,0.0003773966,0.0008260938,0.02135493,0.00572288,0.01016369],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9893126,"threshold_uncertainty_score":0.05652064,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03979580373329172,"score_gpt":0.3027297654122166,"score_spread":0.2629339616789249,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}