{"id":"W4414029349","doi":"10.1101/2025.09.04.670545","title":"Leveraging the largest harmonized epigenomic data collection for metadata prediction validated and augmented over 350,000 public epigenomic datasets","year":2025,"lang":"en","type":"preprint","venue":"bioRxiv (Cold Spring Harbor Laboratory)","topic":"Epigenetics and DNA Methylation","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Sherbrooke","funders":"Fonds de Recherche du Québec - Santé; Natural Sciences and Engineering Research Council of Canada; Fonds de recherche du Québec – Nature et technologies; Alliance de recherche numérique du Canada; Université de Sherbrooke","keywords":"Epigenomics; Metadata; Computer science; Information retrieval; World Wide Web; Biology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01194316,0.001317291,0.001319866,0.004309161,0.001275874,0.002356716,0.002357115,0.0010511,0.002108475],"category_scores_gemma":[0.02217449,0.0007770698,0.001762685,0.003832709,0.0008419405,0.002386159,0.006498923,0.002185739,0.002187506],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008255667,"about_ca_system_score_gemma":0.003172069,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00497354,"about_ca_topic_score_gemma":0.01066151,"domain_scores_codex":[0.9923339,0.0018573,0.0006909511,0.002629405,0.002043656,0.000444801],"domain_scores_gemma":[0.9848925,0.003691466,0.0009732087,0.006156713,0.003761166,0.0005249832],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002394109,0.0009590729,0.2534407,0.004202507,0.003019342,0.001399954,0.00196609,0.06319133,0.1794922,0.01075032,0.1222353,0.3569491],"study_design_scores_gemma":[0.0004643529,0.000847464,0.1984191,0.001097373,0.002179659,0.001534958,0.001332881,0.1279597,0.2480552,0.02631551,0.3910224,0.0007713694],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.3793249,0.005126622,0.2432285,0.001793294,0.0009216195,0.000898838,0.3197673,0.037363,0.011576],"genre_scores_gemma":[0.2050593,0.0009101541,0.2316995,0.001164716,0.0001394909,0.001066119,0.5555179,0.002477526,0.001965395],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9880568,"threshold_uncertainty_score":0.06316221,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03814697571027185,"score_gpt":0.2651204148767424,"score_spread":0.2269734391664705,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}