{"id":"W4388731451","doi":"10.1002/widm.1523","title":"The use of gene expression datasets in feature selection research: 20 years of inherent bias?","year":2023,"lang":"en","type":"article","venue":"Wiley Interdisciplinary Reviews Data Mining and Knowledge Discovery","topic":"Gene expression and cancer classification","field":"Biochemistry, Genetics and Molecular Biology","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"Global Affairs Canada; Fundação de Amparo à Pesquisa do Estado do Rio Grande do Sul; Conselho Nacional de Desenvolvimento Científico e Tecnológico; Coordenação de Aperfeiçoamento de Pessoal de Nível Superior","keywords":"Feature selection; Preprocessor; Computer science; Selection (genetic algorithm); Feature (linguistics); Machine learning; Data mining; DNA microarray; Data pre-processing; Artificial intelligence; Biological data; Data science; Bioinformatics; Gene; Gene expression; Biology; Genetics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.08791871,0.0007651712,0.002211465,0.004587166,0.0006464325,0.005796011,0.002298246,0.001438248,0.001488091],"category_scores_gemma":[0.1577439,0.0005574349,0.001376036,0.01019191,0.003233196,0.005053633,0.00220738,0.003435479,0.0007311601],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002165631,"about_ca_system_score_gemma":0.001805569,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001314566,"about_ca_topic_score_gemma":0.001058543,"domain_scores_codex":[0.9375048,0.03960359,0.005695235,0.005384838,0.01135069,0.0004608154],"domain_scores_gemma":[0.7410305,0.2155561,0.009212708,0.01416514,0.01948095,0.000554604],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0003500228,0.00009477734,0.02551744,0.006507264,0.00211445,0.0002175672,0.0008527301,0.005218522,0.002054708,0.0434922,0.0252972,0.8882832],"study_design_scores_gemma":[0.0003569476,0.0009097595,0.06152196,0.01648054,0.002209036,0.00189454,0.00138211,0.07083129,0.02123703,0.4093029,0.4132797,0.0005941312],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.05007363,0.4809087,0.3949169,0.05511505,0.005388099,0.0002779419,0.002196888,0.0008817927,0.01024104],"genre_scores_gemma":[0.5834417,0.1894115,0.185283,0.02317584,0.01009917,0.001334297,0.003481225,0.0007568593,0.003016288],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9120813,"threshold_uncertainty_score":0.4649642,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2838351985218934,"score_gpt":0.4274522595190564,"score_spread":0.1436170609971629,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}