{"id":"W2981736362","doi":"10.1186/s40246-019-0226-2","title":"Size matters: how sample size affects the reproducibility and specificity of gene set analysis","year":2019,"lang":"en","type":"article","venue":"Human Genomics","topic":"Gene expression and cancer classification","field":"Biochemistry, Genetics and Molecular Biology","cited_by":48,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Sample size determination; Reproducibility; False positive paradox; Sample (material); Set (abstract data type); Data set; Computer science; Statistics; Computational biology; False discovery rate; Data mining; Biology; Gene; Genetics; Artificial intelligence; Mathematics; Chemistry; Chromatography","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1183523,0.000878324,0.001349244,0.001523556,0.001228171,0.002809935,0.001931673,0.002121452,0.001434437],"category_scores_gemma":[0.3011714,0.0006460518,0.001455683,0.001900074,0.003481103,0.002312746,0.001974878,0.001651682,0.0005857833],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001151914,"about_ca_system_score_gemma":0.001516507,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001057135,"about_ca_topic_score_gemma":0.00165721,"domain_scores_codex":[0.845694,0.1028443,0.01275704,0.0124752,0.02499819,0.001231284],"domain_scores_gemma":[0.5089269,0.4244303,0.01296444,0.03349134,0.01933874,0.0008483379],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005262428,0.0008531039,0.2985299,0.004711878,0.005005064,0.001406246,0.004816974,0.02539284,0.2348481,0.01546902,0.0116029,0.3921016],"study_design_scores_gemma":[0.0008028495,0.005078291,0.3103667,0.001939336,0.004499689,0.003952469,0.001511108,0.09119583,0.4715371,0.05962678,0.04869583,0.0007941093],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2877754,0.01125115,0.679059,0.006841665,0.00218186,0.002250725,0.001263921,0.001513074,0.007863227],"genre_scores_gemma":[0.7963526,0.001230225,0.1946908,0.00225042,0.0003379523,0.002232051,0.0008523375,0.0009481016,0.001105495],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8816476,"threshold_uncertainty_score":0.6259145,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02048667910047409,"score_gpt":0.2585244882334746,"score_spread":0.2380378091330005,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}