{"id":"W3150070368","doi":"10.1101/2021.03.27.437353","title":"Analysis of half a billion datapoints across ten machine-learning algorithms identify key determinants of insulin gene transcription","year":2021,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Single-cell and spatial transcriptomics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Impact","funders":"Australian Research Council; National Institutes of Health; Western Sydney University; Novo Nordisk Fonden; Novo Nordisk; Danish Diabetes Academy; Juvenile Diabetes Research Foundation United States of America; Leona M. and Harry B. Helmsley Charitable Trust","keywords":"Random forest; Key (lock); Workflow; Ensemble learning; Gene; Big data; Correlation; Insulin","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.000745518,0.0005207741,0.001048323,0.0003410347,0.0001175096,0.0001095175,0.0006414804,0.0007959934,0.00002215451],"category_scores_gemma":[0.0001910573,0.0005654017,0.0006000038,0.0008790632,0.0001745573,0.00002154225,0.0003843408,0.000502235,0.000001809738],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005559892,"about_ca_system_score_gemma":0.0002840296,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008846812,"about_ca_topic_score_gemma":0.0001764273,"domain_scores_codex":[0.9967032,0.0002575792,0.001035381,0.001134618,0.0004097933,0.0004594096],"domain_scores_gemma":[0.9970878,0.00001988744,0.0007790153,0.001242997,0.0007026233,0.0001676156],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001037155,0.0002639779,0.08121719,0.0003872853,0.001159063,0.0000216119,0.00005454832,0.0007209124,0.9160237,0.000001383356,0.000002075263,0.00004456684],"study_design_scores_gemma":[0.0006216979,0.0001112992,0.1410967,0.0001906442,0.000923317,4.714961e-8,0.00001335014,0.005956303,0.8503836,1.593314e-7,0.0002186357,0.0004841885],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.984928,0.002906751,0.009917379,0.0000116238,0.0005798952,0.000310334,0.001306135,0.00003892288,0.000001020755],"genre_scores_gemma":[0.9927412,0.002199978,0.004661654,0.0000308453,0.0001573693,0.00003770015,0.00006760408,0.00009860866,0.000004988271],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.06564005,"threshold_uncertainty_score":0.9996797,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01876568728954019,"score_gpt":0.2558697607927689,"score_spread":0.2371040735032287,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}