{"meta":{"query_hash":"c7cf3172b93e","filters":{"venue":"BioData Mining"},"cohort_total":48,"direct_labels_cover":0,"predictions_cover":48,"exported":48,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/c7cf3172b93e","api":"https://metacan.xera.ac/api/v1/cohort?venue=BioData+Mining"},"results":[{"id":"W2069849680","doi":"10.1186/1756-0381-6-8","title":"Prediction of Drosophila melanogaster gene function using Support Vector Machines","year":2013,"lang":"en","type":"article","venue":"BioData Mining","topic":"Bioinformatics and Genomic Networks","field":"Biochemistry, Genetics and Molecular Biology","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Public Health Ontario; University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institutes of Health Research; University of Toronto; University of Washington","keywords":"Drosophila melanogaster; Gene; Genome; Biology; Computational biology; Support vector machine; Gene prediction; Genome project; Microarray analysis techniques; Melanogaster; Gene Annotation; Genetics; Annotation; Function (biology); Computer science; Artificial intelligence; Gene expression","score_opus":0.02614048323694971,"score_gpt":0.22492678509502498,"score_spread":0.19878630185807528,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2069849680","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6397695,0.001984334,0.34775883,0.00032142375,0.0001158813,0.00013688156,0.003480259,0.005048724,0.001384261],"genre_scores_gemma":[0.90185684,0.00025536562,0.09251738,0.000062489926,0.00003578119,0.00011998711,0.004411664,0.0000425245,0.00069792406],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995376,0.00011052334,0.000044123848,0.00016270111,0.000098951416,0.000046254096],"domain_scores_gemma":[0.99901426,0.00058769685,0.00008360802,0.00003968486,0.00024284875,0.000031874308],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009452448,0.00089924247,0.0005671892,0.0011401719,0.00020094418,0.0004975952,0.0005206845,0.0005972394,0.0010342686],"category_scores_gemma":[0.0018907278,0.00014299374,0.0007066204,0.0006286465,0.0001473709,0.0003019975,0.00021364604,0.00052420795,0.00063507864],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009248426,0.00042704787,0.04427415,0.0005039807,0.00031587156,0.00035909357,0.00009683138,0.3917468,0.035850126,0.0011199281,0.0063267015,0.5180546],"study_design_scores_gemma":[0.000014900956,0.00007198588,0.004235094,0.000012548353,0.000015469706,0.000053300373,0.00001734499,0.98939556,0.0049817865,0.0007233175,0.00047223273,0.0000064555757],"about_ca_topic_score_codex":0.0022354256,"about_ca_topic_score_gemma":0.001052873,"teacher_disagreement_score":0.0022354256,"about_ca_system_score_codex":0.00040453853,"about_ca_system_score_gemma":0.0004184828,"threshold_uncertainty_score":0.004998982},"labels":[],"label_agreement":null},{"id":"W2101279094","doi":"10.1186/1756-0381-7-17","title":"Computational genetics analysis of grey matter density in Alzheimer’s disease","year":2014,"lang":"en","type":"article","venue":"BioData Mining","topic":"Olfactory and Sensory Function Studies","field":"Neuroscience","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Institute on Aging; National Institute of Biomedical Imaging and Bioengineering; Canadian Institutes of Health Research; National Institutes of Health; Genentech; U.S. National Library of Medicine; IXICO; National Cancer Institute; Servier; Eisai; Northern California Institute for Research and Education; University of California, San Diego; Pfizer; Biogen; BioClinica; F. Hoffmann-La Roche; Synarc; University of Southern California; Medpace; Novartis Pharmaceuticals Corporation; Dartmouth College; U.S. Department of Defense; Eli Lilly and Company; Bristol-Myers Squibb; Alzheimer's Disease Neuroimaging Initiative; National Center for Advancing Translational Sciences; Meso Scale Diagnostics; Alzheimer's Association; Foundation for the National Institutes of Health","keywords":"Grey matter; Disease; Computer science; Data science; Computational biology; Biology; Medicine; Pathology","score_opus":0.19009645677759274,"score_gpt":0.3033707888264548,"score_spread":0.11327433204886206,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2101279094","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8562858,0.00094322715,0.13352822,0.0020403184,0.00008192895,0.000084021456,0.0036482352,0.0020313503,0.0013568125],"genre_scores_gemma":[0.9249869,0.0001481845,0.07085721,0.00023489691,0.00004387165,0.00007530688,0.00312525,0.000077488636,0.00045089045],"study_design_codex":"simulation_or_modeling","study_design_gemma":"observational","domain_scores_codex":[0.99964476,0.00017719199,0.000022993248,0.00008677536,0.00003910388,0.000029192533],"domain_scores_gemma":[0.99608225,0.0034097384,0.00016411484,0.00011110212,0.00013752068,0.00009534222],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015047615,0.00051886524,0.0006442057,0.0020270979,0.00042524197,0.0006425465,0.0009371039,0.0006408201,0.0013435731],"category_scores_gemma":[0.0054075825,0.00027959613,0.0014912566,0.0010595775,0.00036774902,0.00025589784,0.0005557972,0.0006379736,0.0001627448],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00085909036,0.00040581814,0.13410625,0.00022030402,0.001704006,0.00059469783,0.00013591371,0.7861074,0.0028149812,0.0045001465,0.0043964814,0.06415489],"study_design_scores_gemma":[0.00004228361,0.00003379487,0.008567778,0.0000062381887,0.00007468204,0.000058927944,0.000023152606,0.9846427,0.0002917424,0.005974486,0.00027734728,0.0000069374146],"about_ca_topic_score_codex":0.019724369,"about_ca_topic_score_gemma":0.021750053,"teacher_disagreement_score":0.019724369,"about_ca_system_score_codex":0.0009060054,"about_ca_system_score_gemma":0.0013799663,"threshold_uncertainty_score":0.03921914},"labels":[],"label_agreement":null},{"id":"W2102718331","doi":"10.1186/1756-0381-7-25","title":"Updating microbial genomic sequences: improving accuracy &amp; innovation","year":2014,"lang":"en","type":"article","venue":"BioData Mining","topic":"Vibrio bacteria research studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"McMaster University","keywords":"Sanger sequencing; Biology; Genome; Whole genome sequencing; Computational biology; Salmonella enterica; Biovar; Genetics; DNA sequencing; Caulobacter crescentus; Salmonella; Gene; Bacterial protein; Bacteria","score_opus":0.03551480171943568,"score_gpt":0.30131152252589205,"score_spread":0.2657967208064564,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2102718331","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08496558,0.004928005,0.88774174,0.004835231,0.0008460422,0.00031861436,0.0021154757,0.0071301647,0.0071191634],"genre_scores_gemma":[0.12428359,0.0019617027,0.86760587,0.00086406374,0.00024664597,0.00012903441,0.0023860272,0.00071885926,0.0018041588],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98096246,0.0056194835,0.0017871151,0.0029137959,0.008271631,0.0004456358],"domain_scores_gemma":[0.9483619,0.019952573,0.004233462,0.01230311,0.014587835,0.000561013],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021900728,0.0010672046,0.0012773053,0.0035299454,0.0007652704,0.0044480693,0.0034211527,0.002112426,0.0024575978],"category_scores_gemma":[0.07841668,0.0009130259,0.000990984,0.005029032,0.0010949188,0.0060348464,0.0026228016,0.0020329102,0.0035665552],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040852287,0.00017370834,0.021498961,0.0011340961,0.00016706435,0.0003221512,0.0011455974,0.016099913,0.07974458,0.009016027,0.005812893,0.86447656],"study_design_scores_gemma":[0.00018246759,0.0007591658,0.027240263,0.001021279,0.00052136317,0.0027995296,0.0014149945,0.27193096,0.39576527,0.05961592,0.2383621,0.00038670478],"about_ca_topic_score_codex":0.0015533438,"about_ca_topic_score_gemma":0.0019402789,"teacher_disagreement_score":0.021900728,"about_ca_system_score_codex":0.0012181341,"about_ca_system_score_gemma":0.002209012,"threshold_uncertainty_score":0.11582351},"labels":[],"label_agreement":null},{"id":"W2111827030","doi":"10.1186/1756-0381-7-7","title":"Semi-supervised consensus clustering for gene expression data analysis","year":2014,"lang":"en","type":"article","venue":"BioData Mining","topic":"Advanced Clustering Algorithms Research","field":"Computer Science","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"National Research Council Canada","funders":"National Research Council Canada","keywords":"Cluster analysis; Computer science; Data mining; Consensus clustering; Computational biology; Expression (computer science); Artificial intelligence; Biology; Correlation clustering; CURE data clustering algorithm","score_opus":0.10870116120127883,"score_gpt":0.350048260191916,"score_spread":0.24134709899063717,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2111827030","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0048176856,0.0003067981,0.9933073,0.000076289616,0.000028293702,0.00008718965,0.0001320129,0.0009445902,0.00029987356],"genre_scores_gemma":[0.12970492,0.00033533238,0.8670255,0.00009254944,0.00006963529,0.0007530796,0.001086307,0.00025969083,0.0006729325],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9934989,0.0024991746,0.00041448063,0.001443121,0.0019979652,0.00014630263],"domain_scores_gemma":[0.99022293,0.004703601,0.0010365195,0.0012198902,0.0026371626,0.00017986055],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0057240664,0.001321783,0.0023112008,0.00288702,0.0010140119,0.0011492053,0.0027144072,0.001711551,0.0016915504],"category_scores_gemma":[0.014415616,0.0007080265,0.0019309913,0.0031342518,0.0014998583,0.0013623812,0.0012694037,0.002125851,0.0015209311],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032022403,0.0002092676,0.0027090423,0.0010150252,0.00053231395,0.00013991639,0.00041757934,0.657284,0.016900728,0.011743088,0.0056023295,0.30312645],"study_design_scores_gemma":[0.000012550233,0.00003164614,0.0006001565,0.000020787982,0.00001940386,0.00005141592,0.000026929085,0.98296356,0.0040430576,0.010800973,0.0014011542,0.000028431314],"about_ca_topic_score_codex":0.004231688,"about_ca_topic_score_gemma":0.0037342615,"teacher_disagreement_score":0.0057240664,"about_ca_system_score_codex":0.001808852,"about_ca_system_score_gemma":0.0026686862,"threshold_uncertainty_score":0.030272126},"labels":[],"label_agreement":null},{"id":"W2120184682","doi":"10.1186/s13040-014-0032-2","title":"Integrative genomics and transcriptomics analysis of human embryonic and induced pluripotent stem cells","year":2014,"lang":"en","type":"article","venue":"BioData Mining","topic":"Genomic variations and chromosomal abnormalities","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lunenfeld-Tanenbaum Research Institute","funders":"Emil Aaltosen Säätiö; European Commission","keywords":"Embryonic stem cell; Induced pluripotent stem cell; Genomics; Transcriptome; Computational biology; Biology; Human Induced Pluripotent Stem Cells; Stem cell; Data science; Computer science; Genetics; Genome; Gene; Gene expression","score_opus":0.017187619078278094,"score_gpt":0.23010869046941393,"score_spread":0.21292107139113584,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2120184682","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.84226567,0.004438402,0.13107307,0.00019803843,0.000046255096,0.00015901715,0.017268185,0.00094972906,0.0036015487],"genre_scores_gemma":[0.83151996,0.002217734,0.11706608,0.00017426885,0.000036144884,0.00027644177,0.046853345,0.00014031275,0.0017156666],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9997434,0.00003882376,0.000021292542,0.00008443639,0.000088713176,0.000023306111],"domain_scores_gemma":[0.9998883,0.000043290387,0.000023598654,0.000012553246,0.000023961911,0.000008337976],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00038921373,0.00030135326,0.0004171519,0.00074880576,0.0001681479,0.00034527574,0.00024285793,0.00016293935,0.00073569344],"category_scores_gemma":[0.00034289504,0.00010661821,0.00064113346,0.00080983003,0.00014863965,0.00014406026,0.00024803044,0.00020184055,0.00024241768],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021751577,0.00003227163,0.007409539,0.00033159953,0.00009788201,0.00024334743,0.000080788595,0.003495535,0.95644873,0.0014635818,0.00032281308,0.029856442],"study_design_scores_gemma":[0.00005775795,0.00054905855,0.2093904,0.00007632845,0.0005554856,0.0015571144,0.00025399736,0.05804659,0.69658107,0.0061131283,0.026770426,0.00004864509],"about_ca_topic_score_codex":0.00056000205,"about_ca_topic_score_gemma":0.0006925572,"teacher_disagreement_score":0.00074880576,"about_ca_system_score_codex":0.00025958582,"about_ca_system_score_gemma":0.00037504817,"threshold_uncertainty_score":0.0024611354},"labels":[],"label_agreement":null},{"id":"W2136227372","doi":"10.1186/s13040-015-0056-2","title":"The role of visualization and 3-D printing in biological data mining","year":2015,"lang":"en","type":"article","venue":"BioData Mining","topic":"Data Visualization and Analytics","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Institute of General Medical Sciences; National Institute on Aging; National Institute of Biomedical Imaging and Bioengineering; Canadian Institutes of Health Research; National Institutes of Health; Genentech; U.S. National Library of Medicine; IXICO; Servier; Eisai; Northern California Institute for Research and Education; University of California, San Diego; Pfizer; Biogen; BioClinica; F. Hoffmann-La Roche; Synarc; University of Southern California; Medpace; Novartis Pharmaceuticals Corporation; Dartmouth College; U.S. Department of Defense; Eli Lilly and Company; Bristol-Myers Squibb; Alzheimer's Disease Neuroimaging Initiative; National Center for Advancing Translational Sciences; Meso Scale Diagnostics; Alzheimer's Association; Foundation for the National Institutes of Health","keywords":"Visualization; Data science; Computer science; Biological network; Biological data; Data mining; Endophenotype; Data visualization; Artificial intelligence; Machine learning; Bioinformatics; Biology; Neuroscience","score_opus":0.1481506632125918,"score_gpt":0.3576323327334887,"score_spread":0.20948166952089692,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2136227372","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0131034255,0.0025500474,0.9709991,0.003524429,0.00023230936,0.0001014747,0.00050376105,0.0048839473,0.0041015055],"genre_scores_gemma":[0.1612792,0.003901889,0.83164495,0.00051070575,0.00018980126,0.00024949617,0.0004324673,0.0007851042,0.0010063709],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9960742,0.0022590598,0.00034237644,0.00034027203,0.0008902404,0.000093947136],"domain_scores_gemma":[0.9638945,0.028322611,0.0013045662,0.0038272315,0.0020723022,0.00057881686],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00830434,0.0012891294,0.0010155293,0.0033314184,0.00083319296,0.0071402024,0.0018851534,0.0016820233,0.004777777],"category_scores_gemma":[0.027361918,0.0012419358,0.0021658528,0.0028415231,0.002527691,0.003897758,0.0037731507,0.0023894294,0.0011234976],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00088599295,0.00022381477,0.01438052,0.0028312043,0.00060925086,0.0016328668,0.0048325574,0.12913473,0.037497394,0.16687043,0.020019457,0.6210817],"study_design_scores_gemma":[0.00020690437,0.0003429048,0.007409404,0.0010599623,0.0003004881,0.00296186,0.001156751,0.54702425,0.063412346,0.25772026,0.117911346,0.00049360114],"about_ca_topic_score_codex":0.0016704942,"about_ca_topic_score_gemma":0.0009673177,"teacher_disagreement_score":0.00830434,"about_ca_system_score_codex":0.0011809809,"about_ca_system_score_gemma":0.0010038725,"threshold_uncertainty_score":0.043918073},"labels":[],"label_agreement":null},{"id":"W2155858762","doi":"10.1186/1756-0381-4-22","title":"Detection of putative new mutacins by bioinformatic analysis using available web tools","year":2011,"lang":"en","type":"article","venue":"BioData Mining","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Bacteriocin; Computational biology; Context (archaeology); Identification (biology); Genome; Biology; Drug discovery; Whole genome sequencing; Computer science; Genetics; Gene; Bioinformatics; Bacteria","score_opus":0.06311613741850325,"score_gpt":0.2527864325412549,"score_spread":0.18967029512275163,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2155858762","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.88804877,0.0033361171,0.09230833,0.0005597875,0.000052352014,0.00033140657,0.0070755747,0.004933505,0.0033541047],"genre_scores_gemma":[0.7312779,0.0015436502,0.24163535,0.00022289495,0.000043444114,0.0003565066,0.02269402,0.0004166081,0.0018097064],"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","domain_scores_codex":[0.9998073,0.00003402587,0.000028838796,0.00003824902,0.000073069496,0.00001857034],"domain_scores_gemma":[0.999481,0.00020345645,0.00012555598,0.000036683665,0.00007629369,0.00007707995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00048178725,0.0007384148,0.00068304036,0.002135197,0.00041638492,0.0009353863,0.00046418363,0.00037811432,0.0024245288],"category_scores_gemma":[0.0010672362,0.00021161562,0.00074004644,0.0013092484,0.00021766436,0.0006991329,0.0004438385,0.0004955912,0.00077631994],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022770078,0.0006453522,0.027179841,0.002078881,0.00030161266,0.0020638963,0.00034272007,0.0039742864,0.8302581,0.001992469,0.0014098409,0.12747608],"study_design_scores_gemma":[0.0002960618,0.0012870088,0.12739123,0.00028241015,0.00094856677,0.0073810336,0.000910189,0.14985113,0.66779524,0.008531883,0.03515875,0.00016649268],"about_ca_topic_score_codex":0.000278276,"about_ca_topic_score_gemma":0.0004530542,"teacher_disagreement_score":0.0024245288,"about_ca_system_score_codex":0.00029001923,"about_ca_system_score_gemma":0.00035167506,"threshold_uncertainty_score":0.008110881},"labels":[],"label_agreement":null},{"id":"W2161488150","doi":"10.1186/1756-0381-5-8","title":"A comparison and evaluation of five biclustering algorithms by quantifying goodness of biclusters for gene expression data","year":2012,"lang":"en","type":"article","venue":"BioData Mining","topic":"Gene expression and cancer classification","field":"Biochemistry, Genetics and Molecular Biology","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"York University","keywords":"Biclustering; Ranking (information retrieval); Computer science; Data mining; Gene ontology; Correlation; Algorithm; Expression (computer science); Cluster analysis; Artificial intelligence; Pattern recognition (psychology); Mathematics; Gene; Gene expression; Biology; Correlation clustering","score_opus":0.26423594520003696,"score_gpt":0.4307928668063278,"score_spread":0.16655692160629082,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2161488150","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.38906106,0.003506394,0.5978411,0.00035094112,0.000147273,0.0005146197,0.001299096,0.0052666455,0.0020129653],"genre_scores_gemma":[0.47887847,0.0007222173,0.51543796,0.00015119252,0.000032824304,0.0005233664,0.0035262352,0.0003482427,0.00037950644],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99478614,0.0019461523,0.000673015,0.0008864629,0.0014347757,0.00027340744],"domain_scores_gemma":[0.98368895,0.01025454,0.0011434982,0.0011310258,0.0034018985,0.00038026113],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010654592,0.0017293486,0.0014663131,0.006933718,0.0011244854,0.0016876535,0.0013653096,0.0011227251,0.00075647817],"category_scores_gemma":[0.021584637,0.0004049869,0.0018649804,0.0036622623,0.0008283704,0.0014482313,0.0012150152,0.0010138766,0.00036782515],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023637025,0.0006190658,0.11310517,0.0015691511,0.0020797371,0.00017533674,0.000744616,0.27685267,0.022787115,0.00419135,0.004626405,0.57088566],"study_design_scores_gemma":[0.000102835824,0.00070880493,0.022758689,0.00011020927,0.00033078756,0.00029734347,0.00042653034,0.9523413,0.017598815,0.003240008,0.0019899195,0.000094752526],"about_ca_topic_score_codex":0.0030686683,"about_ca_topic_score_gemma":0.0029364298,"teacher_disagreement_score":0.010654592,"about_ca_system_score_codex":0.0011653285,"about_ca_system_score_gemma":0.0017951091,"threshold_uncertainty_score":0.05634755},"labels":[],"label_agreement":null},{"id":"W2165336562","doi":"10.1186/1756-0381-5-14","title":"Peer2ref: a peer-reviewer finding web tool that uses author disambiguation","year":2012,"lang":"en","type":"article","venue":"BioData Mining","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ottawa Hospital","funders":"","keywords":"Computer science; MEDLINE; Information retrieval; Web of science; World Wide Web; Selection (genetic algorithm); Data science; Artificial intelligence","score_opus":0.11928096650161898,"score_gpt":0.35438555593076476,"score_spread":0.23510458942914578,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2165336562","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008359505,0.0025768478,0.6225388,0.0027538764,0.0024369925,0.003578244,0.0436114,0.29685122,0.017293163],"genre_scores_gemma":[0.02903231,0.001336534,0.89727974,0.0007546177,0.001304763,0.0023263409,0.035858165,0.017180827,0.014926655],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9847112,0.0042964355,0.0031951822,0.0024624118,0.004904316,0.0004304615],"domain_scores_gemma":[0.88454646,0.05935724,0.012727747,0.012227881,0.026677586,0.0044630254],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.029585812,0.0026977742,0.0026570407,0.033184923,0.003908306,0.007577324,0.0039202473,0.0032650104,0.056233842],"category_scores_gemma":[0.0927793,0.0018520666,0.0014145694,0.01518847,0.0012672003,0.011008887,0.0065397206,0.0016665538,0.045649014],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00090048235,0.0002698904,0.0069876355,0.006407685,0.00039216093,0.0009537725,0.0020089068,0.001044661,0.012126841,0.010011188,0.38833225,0.5705645],"study_design_scores_gemma":[0.0005281137,0.00026436764,0.0070669292,0.0013379657,0.00029726262,0.0018224155,0.0010998503,0.024359163,0.029721823,0.022562288,0.91035837,0.00058148656],"about_ca_topic_score_codex":0.0015823059,"about_ca_topic_score_gemma":0.0028074342,"teacher_disagreement_score":0.97041416,"about_ca_system_score_codex":0.0010734668,"about_ca_system_score_gemma":0.008008891,"threshold_uncertainty_score":0.1881209},"labels":[],"label_agreement":null},{"id":"W2171815584","doi":"10.1186/1756-0381-7-22","title":"Applications of the MapReduce programming framework to clinical big data analysis: current landscape and future trends","year":2014,"lang":"en","type":"review","venue":"BioData Mining","topic":"Artificial Intelligence in Healthcare","field":"Health Professions","cited_by":147,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Calgary Laboratory Services; University of Calgary","funders":"","keywords":"Computer science; Big data; Scalability; Programming paradigm; Data-intensive computing; Distributed computing; Data processing; Node (physics); Distributed File System; Grid computing; Grid; Database; Data mining; Operating system; Programming language","score_opus":0.5193724130788084,"score_gpt":0.6095661319612116,"score_spread":0.09019371888240313,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2171815584","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0047118533,0.8185324,0.11895419,0.033237852,0.0028795374,0.00024145596,0.0002460678,0.001315565,0.019881023],"genre_scores_gemma":[0.026760148,0.84430104,0.111452326,0.006766128,0.006480271,0.0002045863,0.0005551095,0.00035504217,0.0031253293],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.997127,0.00080060627,0.00023627856,0.0004828221,0.0011601048,0.0001931875],"domain_scores_gemma":[0.9915741,0.004479284,0.0003651902,0.00034944568,0.002591527,0.0006405295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0072562657,0.0007438996,0.00092666864,0.0027752947,0.00057909294,0.0035711962,0.0030910405,0.0017437913,0.0023321572],"category_scores_gemma":[0.007740624,0.0007642454,0.001118306,0.004328601,0.0022616228,0.0051590847,0.002272269,0.004667703,0.0012571561],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014198683,0.00024176002,0.0036925804,0.0064220694,0.00011299608,0.00026171698,0.0006658817,0.004262401,0.0021302626,0.04976753,0.047156096,0.88514465],"study_design_scores_gemma":[0.000050569866,0.00031566978,0.003609128,0.0031631049,0.000098076096,0.0017387796,0.0009926399,0.016245501,0.0028554052,0.07175822,0.89897114,0.00020169599],"about_ca_topic_score_codex":0.0035041128,"about_ca_topic_score_gemma":0.0025070915,"teacher_disagreement_score":0.0072562657,"about_ca_system_score_codex":0.0017959465,"about_ca_system_score_gemma":0.0038722942,"threshold_uncertainty_score":0.03837526},"labels":[],"label_agreement":null},{"id":"W2202714662","doi":"10.1186/s13040-015-0077-x","title":"Characterizing gene-gene interactions in a statistical epistasis network of twelve candidate genes for obesity","year":2015,"lang":"en","type":"article","venue":"BioData Mining","topic":"Genetic Associations and Epidemiology","field":"Biochemistry, Genetics and Molecular Biology","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"U.S. National Library of Medicine; National Institute of General Medical Sciences; National Institutes of Health","keywords":"Epistasis; SNP; Missing heritability problem; Single-nucleotide polymorphism; Genetics; Gene; Genome-wide association study; Biology; Candidate gene; Pairwise comparison; Computational biology; Statistics; Genotype; Mathematics","score_opus":0.059058872348792146,"score_gpt":0.32741833500269707,"score_spread":0.2683594626539049,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2202714662","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97900873,0.00024101313,0.019735882,0.00008640123,0.000002931369,0.000026072894,0.0003916768,0.000075396725,0.0004319378],"genre_scores_gemma":[0.9942993,0.00005925081,0.0051170886,0.00001454071,0.000004521028,0.000027991839,0.00039908837,0.000007123974,0.00007108158],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9994185,0.00024026803,0.00003302899,0.0001794842,0.00007197949,0.000056777466],"domain_scores_gemma":[0.9963864,0.0025067045,0.00067421026,0.00014110006,0.0001415289,0.00015003016],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009455682,0.00038187136,0.0004107569,0.0021187146,0.000404875,0.0005184203,0.0003399827,0.0002657928,0.0012079478],"category_scores_gemma":[0.0044446685,0.00020298483,0.0007105871,0.0016739602,0.0004974178,0.00041196274,0.00061114423,0.0004795095,0.00010001097],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008212848,0.00021084033,0.8003345,0.00038268723,0.0024295754,0.000998456,0.0005427947,0.096029,0.042231236,0.0055670235,0.00096738886,0.04948519],"study_design_scores_gemma":[0.000056640838,0.000285555,0.6152518,0.000025651272,0.00063604926,0.0009552566,0.00031411755,0.36022475,0.003565488,0.017524736,0.0011144939,0.00004549523],"about_ca_topic_score_codex":0.0023630194,"about_ca_topic_score_gemma":0.0036059942,"teacher_disagreement_score":0.0023630194,"about_ca_system_score_codex":0.0004528728,"about_ca_system_score_gemma":0.00038460744,"threshold_uncertainty_score":0.0050007105},"labels":[],"label_agreement":null},{"id":"W2222833576","doi":"10.1186/s13040-015-0062-4","title":"Functional dyadicity and heterophilicity of gene-gene interactions in statistical epistasis networks","year":2015,"lang":"en","type":"article","venue":"BioData Mining","topic":"Bioinformatics and Genomic Networks","field":"Biochemistry, Genetics and Molecular Biology","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"National Institute of Allergy and Infectious Diseases; U.S. National Library of Medicine; National Institute of General Medical Sciences; National Cancer Institute; National Institutes of Health","keywords":"Epistasis; Context (archaeology); Computational biology; Biology; Genetic association; Gene; Gene interaction; Gene ontology; Genetics; Computer science; Genotype; Single-nucleotide polymorphism; Gene expression","score_opus":0.04534004385984268,"score_gpt":0.2740099323237019,"score_spread":0.22866988846385922,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2222833576","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7110706,0.00066615385,0.285111,0.0003381478,0.000010384719,0.000052334184,0.00076081516,0.00021402731,0.0017765153],"genre_scores_gemma":[0.9880144,0.00012725189,0.011306724,0.000024902321,0.000011722071,0.00004078366,0.0003003288,0.000011905496,0.00016190989],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99900454,0.00043626517,0.000056985045,0.00030942296,0.00012740737,0.000065376],"domain_scores_gemma":[0.9929141,0.0047254153,0.0014492434,0.0004045986,0.00028428977,0.00022233144],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013979592,0.00030705964,0.00043561312,0.0027725815,0.0004942809,0.0008968103,0.00047609454,0.00042121878,0.00095032016],"category_scores_gemma":[0.0072056456,0.000185863,0.0006888086,0.0018717938,0.0010601378,0.0010119631,0.00077829545,0.00049555954,0.00010528731],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000555155,0.00026540222,0.35695583,0.00073993806,0.0011087937,0.0010287395,0.0016202646,0.3640315,0.04044947,0.13873419,0.0024731106,0.09203753],"study_design_scores_gemma":[0.000023237046,0.00009496093,0.10147138,0.000032128988,0.00019357441,0.0005746216,0.00035767097,0.7661652,0.0029898428,0.12634346,0.0017165572,0.000037345744],"about_ca_topic_score_codex":0.0026483047,"about_ca_topic_score_gemma":0.0031064763,"teacher_disagreement_score":0.0027725815,"about_ca_system_score_codex":0.00077137665,"about_ca_system_score_gemma":0.00040422747,"threshold_uncertainty_score":0.0073931813},"labels":[],"label_agreement":null},{"id":"W2232773008","doi":"10.1186/s13040-015-0078-9","title":"Iteratively refining breast cancer intrinsic subtypes in the METABRIC dataset","year":2016,"lang":"en","type":"article","venue":"BioData Mining","topic":"AI in cancer detection","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"BC Cancer Agency; Australian Research Council; Cancer Research UK","keywords":"Discriminative model; Computer science; Machine learning; Artificial intelligence; Set (abstract data type); Breast cancer; Class (philosophy); Data mining; Pattern recognition (psychology); Cancer; Medicine","score_opus":0.04386414633269161,"score_gpt":0.2884951767383135,"score_spread":0.24463103040562187,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2232773008","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8779627,0.0014966648,0.10512367,0.00050061115,0.000046204877,0.00030902468,0.0103004165,0.002293804,0.0019670362],"genre_scores_gemma":[0.70352566,0.00035240545,0.2502988,0.00031768027,0.00004490453,0.00031475569,0.043477323,0.00025558774,0.0014129265],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99816245,0.0006053101,0.00011948364,0.0006069587,0.00032419572,0.00018155562],"domain_scores_gemma":[0.9960304,0.0022511207,0.00028380673,0.0006213362,0.00068942324,0.00012395748],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0053265058,0.0010241913,0.0011858095,0.0027213967,0.000836023,0.0013521449,0.0014120492,0.0010452544,0.00078605046],"category_scores_gemma":[0.008282722,0.0002645479,0.0012969798,0.0015477608,0.00046603862,0.0004949331,0.001174288,0.0014732883,0.0007205757],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024507833,0.0006065761,0.39396864,0.0007914436,0.0010869722,0.0005816709,0.0010187741,0.15174837,0.04121358,0.002454781,0.015531177,0.38854727],"study_design_scores_gemma":[0.00021780551,0.0004860929,0.1315217,0.0001335005,0.0006065757,0.0009318055,0.0006940782,0.7963581,0.040715154,0.008584756,0.01962989,0.0001205267],"about_ca_topic_score_codex":0.010217408,"about_ca_topic_score_gemma":0.020695666,"teacher_disagreement_score":0.010217408,"about_ca_system_score_codex":0.0009698545,"about_ca_system_score_gemma":0.0013445466,"threshold_uncertainty_score":0.028169632},"labels":[],"label_agreement":null},{"id":"W2273538503","doi":"10.1186/s13040-016-0082-8","title":"Network-based analysis of genetic variants associated with hippocampal volume in Alzheimer’s disease: a study of ADNI cohorts","year":2016,"lang":"en","type":"article","venue":"BioData Mining","topic":"Genetic Associations and Epidemiology","field":"Biochemistry, Genetics and Molecular Biology","cited_by":35,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Institute of General Medical Sciences; National Institute on Aging; National Institute of Biomedical Imaging and Bioengineering; Canadian Institutes of Health Research; University of California, San Diego; National Institutes of Health; Genentech; U.S. National Library of Medicine; IXICO; Servier; Eisai; Pfizer; Biogen; BioClinica; Medpace; F. Hoffmann-La Roche; Synarc; University of Southern California; Novartis Pharmaceuticals Corporation; Dartmouth College; U.S. Department of Defense; Eli Lilly and Company; Bristol-Myers Squibb; Alzheimer's Disease Neuroimaging Initiative; National Center for Advancing Translational Sciences; Meso Scale Diagnostics; Gordon and Betty Moore Foundation; Foundation for the National Institutes of Health; Alzheimer's Association; National Science Foundation","keywords":"Hippocampal formation; Disease; Volume (thermodynamics); Genetic variants; Alzheimer's disease; Neuroscience; Computer science; Medicine; Data science; Computational biology; Biology; Internal medicine; Genetics; Genotype; Gene","score_opus":0.028979905395350163,"score_gpt":0.2713402144407311,"score_spread":0.24236030904538092,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2273538503","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98405313,0.0006178373,0.0034323065,0.00028826323,0.000032865155,0.000069894086,0.010288793,0.00011924473,0.0010978108],"genre_scores_gemma":[0.9749919,0.00033814256,0.006659483,0.00012675788,0.000042756947,0.00022306759,0.016890202,0.00008475594,0.00064302835],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99897504,0.00028862414,0.0000686734,0.00047873345,0.00011518521,0.00007377515],"domain_scores_gemma":[0.9971973,0.0007639489,0.0007786213,0.0006270112,0.00035005176,0.00028307654],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002057068,0.00051173964,0.0005412117,0.001982731,0.00091063464,0.0009555484,0.0010485806,0.0005446528,0.0018341847],"category_scores_gemma":[0.0075380565,0.00036586492,0.00082577317,0.0019607486,0.00041881844,0.0005836795,0.0011403863,0.00065733783,0.00034205287],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014108822,0.000115591916,0.9729864,0.00016930433,0.0030950955,0.000448463,0.00047583072,0.0030305712,0.0020712074,0.0008855816,0.0054218173,0.009889365],"study_design_scores_gemma":[0.00017634085,0.00012746398,0.9840792,0.00005219172,0.0011927251,0.0008125309,0.0002468229,0.008330313,0.00056674244,0.002073373,0.002301518,0.000040761115],"about_ca_topic_score_codex":0.014758114,"about_ca_topic_score_gemma":0.026619181,"teacher_disagreement_score":0.014758114,"about_ca_system_score_codex":0.00082742603,"about_ca_system_score_gemma":0.0006509924,"threshold_uncertainty_score":0.02934444},"labels":[],"label_agreement":null},{"id":"W2368906419","doi":"10.1186/s13040-016-0098-0","title":"Machine learning algorithms for mode-of-action classification in toxicity assessment","year":2016,"lang":"en","type":"article","venue":"BioData Mining","topic":"Gene expression and cancer classification","field":"Biochemistry, Genetics and Molecular Biology","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"MacEwan University; University of Calgary; Alberta Health; University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs","keywords":"Computer science; Action (physics); Mode (computer interface); Machine learning; Artificial intelligence; Mode of action; Algorithm; Chemistry; Human–computer interaction","score_opus":0.08482953243657332,"score_gpt":0.3714903720574589,"score_spread":0.28666083962088557,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2368906419","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0071478537,0.001182267,0.9899238,0.00018828547,0.000050868046,0.00006919301,0.00010079849,0.0006904341,0.0006465438],"genre_scores_gemma":[0.2604237,0.0017249899,0.73446196,0.00017910928,0.0002085995,0.00062529655,0.00059075834,0.00011737741,0.0016682551],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998547,0.0005572566,0.00017056742,0.0002576798,0.00039370154,0.00007379047],"domain_scores_gemma":[0.9957841,0.0027882967,0.0004296815,0.0002337194,0.00070630206,0.000057885703],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028425502,0.0011684217,0.0014320803,0.0019048505,0.000412464,0.0010290287,0.0013693665,0.0014437085,0.0015078513],"category_scores_gemma":[0.0064611603,0.00036316103,0.0010236148,0.0018600355,0.0005806075,0.0010390932,0.0007119712,0.0021687895,0.00084741553],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013217785,0.00023084115,0.0033614843,0.00041583795,0.00019981743,0.00008859761,0.000083126,0.461932,0.00562706,0.009469985,0.0028507132,0.5156084],"study_design_scores_gemma":[0.000004790406,0.000040460185,0.00035692521,0.000017011314,0.00001074266,0.000022652706,0.0000074727345,0.9926025,0.0012306588,0.0049470747,0.0007504988,0.0000091785605],"about_ca_topic_score_codex":0.0014517059,"about_ca_topic_score_gemma":0.00082518125,"teacher_disagreement_score":0.0028425502,"about_ca_system_score_codex":0.0007787796,"about_ca_system_score_gemma":0.0007911481,"threshold_uncertainty_score":0.015033007},"labels":[],"label_agreement":null},{"id":"W2528021625","doi":"10.1186/s13040-017-0129-5","title":"Variant Set Enrichment: an R package to identify disease-associated functional genomic regions","year":2017,"lang":"en","type":"article","venue":"BioData Mining","topic":"Genomics and Rare Diseases","field":"Biochemistry, Genetics and Molecular Biology","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University Health Network; Ontario Institute for Cancer Research; University of Toronto; Princess Margaret Cancer Centre","funders":"U.S. National Library of Medicine; National Institute of Environmental Health Sciences; Canadian Cancer Society Research Institute; Canadian Institutes of Health Research; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; National Cancer Institute; National Institutes of Health; Prostate Cancer Canada; Natural Sciences and Engineering Research Council of Canada; Princess Margaret Cancer Foundation","keywords":"Computational biology; Disease; Set (abstract data type); Genome; R package; Human genome; Computer science; Biology; Genetics; Gene; Medicine","score_opus":0.05028613776478923,"score_gpt":0.31212661942188835,"score_spread":0.2618404816570991,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2528021625","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0153005645,0.0021127795,0.64047694,0.0013517708,0.00068082986,0.0005069182,0.12112073,0.21413225,0.004317294],"genre_scores_gemma":[0.11311482,0.0014504362,0.70608693,0.0013846895,0.00031910648,0.0034509727,0.08710953,0.082539365,0.0045440462],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961591,0.0019015237,0.00032883702,0.00089004636,0.00052760664,0.00019284873],"domain_scores_gemma":[0.9748155,0.020943208,0.0011764237,0.0018949199,0.00075774064,0.00041213166],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006543024,0.003237608,0.0025138536,0.003845889,0.00080361904,0.0024927978,0.0038675386,0.0012003542,0.031513743],"category_scores_gemma":[0.04748147,0.0015682284,0.004023466,0.0035861267,0.0010674547,0.0014361029,0.0036062405,0.0026617607,0.015931677],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0034602548,0.00023817013,0.058917806,0.009620402,0.016033776,0.0033459095,0.0013199267,0.035172615,0.01328803,0.032887705,0.6030368,0.22267856],"study_design_scores_gemma":[0.0021745975,0.0005969789,0.04357469,0.0010775791,0.0059099393,0.0055332948,0.00040350956,0.20479286,0.02274013,0.14447977,0.56797683,0.0007397372],"about_ca_topic_score_codex":0.0026831694,"about_ca_topic_score_gemma":0.002939774,"teacher_disagreement_score":0.031513743,"about_ca_system_score_codex":0.0005844833,"about_ca_system_score_gemma":0.0027433007,"threshold_uncertainty_score":0.10542399},"labels":[],"label_agreement":null},{"id":"W2587099845","doi":"10.1186/s13040-017-0123-y","title":"Semantics-based plausible reasoning to extend the knowledge coverage of medical knowledge bases for improved clinical decision support","year":2017,"lang":"en","type":"article","venue":"BioData Mining","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Model-based reasoning; Semantics (computer science); Semantic Web; Data science; Artificial intelligence; Information retrieval; Knowledge representation and reasoning","score_opus":0.076125745526062,"score_gpt":0.4277109479327042,"score_spread":0.3515852024066422,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2587099845","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025323633,0.0013276209,0.96383095,0.002332294,0.00007594311,0.00050194375,0.0017298489,0.0021641143,0.0027136982],"genre_scores_gemma":[0.29605663,0.0008023266,0.6975414,0.0007907345,0.00011196602,0.0003284332,0.0038247544,0.0001389423,0.00040477168],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.984385,0.006653038,0.002535575,0.0020387631,0.004039368,0.0003483242],"domain_scores_gemma":[0.95116884,0.037522227,0.0023324566,0.0049690194,0.0034649605,0.00054251315],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016976617,0.001452508,0.0016331233,0.008250455,0.0011471983,0.004963693,0.0031732935,0.0018535331,0.0031014432],"category_scores_gemma":[0.074995145,0.0008898011,0.0042814417,0.0052235117,0.0018738466,0.009463459,0.005553761,0.0026840128,0.000719651],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010876284,0.0010531265,0.0148470355,0.0027050667,0.0013373173,0.0031093082,0.0037388974,0.31359756,0.012715942,0.12758754,0.009137814,0.5090828],"study_design_scores_gemma":[0.0001967267,0.00015438178,0.0014684091,0.00039494794,0.00041054067,0.0007503456,0.00042858583,0.7074946,0.008411471,0.26583326,0.014343642,0.00011299068],"about_ca_topic_score_codex":0.0049434076,"about_ca_topic_score_gemma":0.006416947,"teacher_disagreement_score":0.016976617,"about_ca_system_score_codex":0.0017239585,"about_ca_system_score_gemma":0.0036418682,"threshold_uncertainty_score":0.08978194},"labels":[],"label_agreement":null},{"id":"W2610597045","doi":"10.1186/s13040-017-0136-6","title":"Study of Meta-analysis strategies for network inference using information-theoretic approaches","year":2017,"lang":"en","type":"article","venue":"BioData Mining","topic":"Gene Regulatory Network Analysis","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Princess Margaret Cancer Centre; University of Toronto; Ontario Institute for Cancer Research","funders":"Université de Liège; Fonds De La Recherche Scientifique - FNRS","keywords":"Pairwise comparison; Computer science; Data mining; Inference; Set (abstract data type); Reverse engineering; Machine learning; Artificial intelligence","score_opus":0.21284763766452527,"score_gpt":0.3422631988385581,"score_spread":0.12941556117403283,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2610597045","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008237532,0.001999553,0.9878896,0.0006148358,0.000065369895,0.00013451719,0.00025230562,0.0002739744,0.00053229986],"genre_scores_gemma":[0.2230891,0.0017289596,0.772069,0.00037444223,0.0002722681,0.00080857385,0.00086538185,0.00028698545,0.000505364],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9780586,0.017756037,0.00062108313,0.0017905064,0.0015092292,0.0002646213],"domain_scores_gemma":[0.7937545,0.19098163,0.0044296472,0.0058342204,0.003941426,0.0010585441],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.044122685,0.003540511,0.0028146585,0.008384036,0.0010263798,0.003635593,0.0038768416,0.0022396422,0.0024621214],"category_scores_gemma":[0.09131341,0.0013085913,0.007484455,0.0033900652,0.0014193726,0.0049075475,0.0029418208,0.0043085306,0.00040778084],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041254496,0.00027407572,0.012363986,0.0015313221,0.01095192,0.0003646077,0.0003621189,0.774848,0.0019731757,0.082141526,0.001804673,0.11297202],"study_design_scores_gemma":[0.000043216427,0.0001299463,0.00054697914,0.00014125892,0.00083838735,0.00007313789,0.000051463256,0.9275525,0.00085310114,0.0686865,0.0010546606,0.000028752484],"about_ca_topic_score_codex":0.0027974145,"about_ca_topic_score_gemma":0.0029553797,"teacher_disagreement_score":0.044122685,"about_ca_system_score_codex":0.0026526116,"about_ca_system_score_gemma":0.0028665562,"threshold_uncertainty_score":0.2333458},"labels":[],"label_agreement":null},{"id":"W2736939530","doi":"10.1186/s13040-017-0145-5","title":"Discovery and replication of SNP-SNP interactions for quantitative lipid traits in over 60,000 individuals","year":2017,"lang":"en","type":"article","venue":"BioData Mining","topic":"Genetic Associations and Epidemiology","field":"Biochemistry, Genetics and Molecular Biology","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"U.S. National Library of Medicine; National Center for Advancing Translational Sciences; National Human Genome Research Institute; National Heart, Lung, and Blood Institute; National Eye Institute; National Institute on Aging; Medical Research Council; British Heart Foundation; European Hematology Association; National Institute of General Medical Sciences; National Institute for Health and Care Research; Broad Institute; Harvard University; National Institutes of Health; U.S. Department of Health and Human Services","keywords":"Replication (statistics); SNP; Single-nucleotide polymorphism; Computational biology; Biology; Pairwise comparison; Genetics; Bioinformatics; Gene; Computer science; Artificial intelligence; Genotype","score_opus":0.09098401409320957,"score_gpt":0.3887127870929476,"score_spread":0.29772877299973804,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2736939530","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.86172044,0.0015626411,0.11621792,0.00038068736,0.00021330344,0.0009374394,0.015936594,0.0015236429,0.001507402],"genre_scores_gemma":[0.9154226,0.00025725274,0.06206744,0.00018321749,0.000053712054,0.0012009138,0.019468917,0.0002926618,0.0010532942],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9847853,0.0045883693,0.0017517362,0.005903948,0.0020522191,0.00091842195],"domain_scores_gemma":[0.97005075,0.014340069,0.0022195817,0.010809958,0.0018927457,0.0006868132],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.029355582,0.0012865425,0.002107687,0.001838289,0.0017370675,0.0016671084,0.0018141839,0.0011838254,0.0029960107],"category_scores_gemma":[0.050247878,0.00087131007,0.0036031373,0.003015279,0.00082898786,0.00064212776,0.0020973254,0.0015579647,0.0009508346],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021711064,0.0006412789,0.8570519,0.0006341978,0.007182675,0.0026349488,0.0014097559,0.00715772,0.043998268,0.0016214326,0.0050436216,0.07045312],"study_design_scores_gemma":[0.0015203034,0.001604005,0.90766853,0.0001322665,0.0071910475,0.0026326284,0.000405197,0.03126699,0.02288746,0.0062356996,0.018234076,0.00022179514],"about_ca_topic_score_codex":0.009470483,"about_ca_topic_score_gemma":0.01328111,"teacher_disagreement_score":0.029355582,"about_ca_system_score_codex":0.0005196262,"about_ca_system_score_gemma":0.0023808165,"threshold_uncertainty_score":0.15524906},"labels":[],"label_agreement":null},{"id":"W2771169143","doi":"10.1186/s13040-017-0155-3","title":"Ten quick tips for machine learning in computational biology","year":2017,"lang":"en","type":"review","venue":"BioData Mining","topic":"Genetics, Bioinformatics, and Biomedical Research","field":"Biochemistry, Genetics and Molecular Biology","cited_by":915,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Princess Margaret Cancer Centre","funders":"Natural Sciences and Engineering Research Council of Canada; University of California, Irvine; University of Toronto","keywords":"Computer science; Informatics; Context (archaeology); Data science; Artificial intelligence; Machine learning; Biology; Engineering","score_opus":0.14248121591791127,"score_gpt":0.4167059229229167,"score_spread":0.27422470700500545,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2771169143","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00010513473,0.9722615,0.0036301452,0.014171556,0.0068238107,0.000022274424,0.000059887287,0.0001094646,0.002816182],"genre_scores_gemma":[0.0013540853,0.97440434,0.0071552317,0.009307989,0.00441902,0.00007070455,0.00014040896,0.00006970636,0.0030784486],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.997595,0.00076335913,0.0003634931,0.00024674114,0.00088742934,0.00014391774],"domain_scores_gemma":[0.9837859,0.011665455,0.00064935896,0.00043195064,0.0027518584,0.00071540376],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005610542,0.0015925666,0.0013153126,0.005336689,0.001058515,0.0034549022,0.0021081655,0.0032961017,0.006799325],"category_scores_gemma":[0.017465299,0.00064025447,0.0010108164,0.0051521026,0.0032183714,0.008845662,0.0030435142,0.011145727,0.005549935],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000054409662,0.00006481427,0.00021320781,0.011062316,0.000081704675,0.00027164957,0.00038945637,0.00045387482,0.00050454895,0.034891345,0.25155398,0.70045865],"study_design_scores_gemma":[0.000009507223,0.000031133946,0.00015336201,0.005725003,0.000027688622,0.0004726915,0.00011584555,0.0000912203,0.00015562994,0.018686568,0.97450215,0.000029286766],"about_ca_topic_score_codex":0.0010712193,"about_ca_topic_score_gemma":0.0021038225,"teacher_disagreement_score":0.006799325,"about_ca_system_score_codex":0.0017194549,"about_ca_system_score_gemma":0.0040825675,"threshold_uncertainty_score":0.029671729},"labels":[],"label_agreement":null},{"id":"W3040486170","doi":"10.1186/s13040-020-00218-7","title":"An epistatic interaction between pre-natal smoke exposure and socioeconomic status has a significant impact on bronchodilator drug response in African American youth with asthma","year":2020,"lang":"en","type":"article","venue":"BioData Mining","topic":"Health, Environment, Cognitive Aging","field":"Environmental Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"National Institute of General Medical Sciences; University of California, San Francisco; National Institutes of Health; National Institute of Environmental Health Sciences; San Francisco State University; National Institute on Minority Health and Health Disparities; Tobacco-Related Disease Research Program; National Heart, Lung, and Blood Institute; Gordon and Betty Moore Foundation; American Asthma Foundation; Alfred P. Sloan Foundation; Robert Wood Johnson Foundation","keywords":"Bronchodilator; Socioeconomic status; Asthma; Smoke; Drug; Drug response; Environmental health; African american; Medicine; Demography; Pharmacology; Internal medicine; Geography; Sociology","score_opus":0.036390835657297366,"score_gpt":0.29049632596453406,"score_spread":0.2541054903072367,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3040486170","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9942405,0.0003774319,0.0008578554,0.00018738346,0.000012157216,0.000008824131,0.003885128,0.00004380663,0.00038668266],"genre_scores_gemma":[0.9970976,0.000082952574,0.000778972,0.000035314166,0.0000072952225,0.000015800371,0.0018573897,0.00000768571,0.00011696225],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9990778,0.00037963447,0.00006572036,0.00024501912,0.00008636786,0.00014545978],"domain_scores_gemma":[0.99807286,0.0011354235,0.00038665318,0.00014624355,0.00011272452,0.00014612575],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011265648,0.0004192059,0.000516975,0.0007546019,0.00070772535,0.0005316465,0.00042659178,0.00032497017,0.0034890817],"category_scores_gemma":[0.0048194192,0.00015352521,0.0016136265,0.0010333557,0.00027517232,0.00031555872,0.00086521846,0.0006205536,0.000167435],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00034822407,0.000030154031,0.9891278,0.000040089268,0.00047737916,0.00016290827,0.00012893572,0.0009964616,0.0009993371,0.0001363098,0.0007996526,0.0067528086],"study_design_scores_gemma":[0.000017891423,0.00010167734,0.98812675,0.000028580707,0.00035906525,0.0003007892,0.00037235036,0.008849788,0.00039622412,0.00049815746,0.000934199,0.000014565168],"about_ca_topic_score_codex":0.022465676,"about_ca_topic_score_gemma":0.03666135,"teacher_disagreement_score":0.022465676,"about_ca_system_score_codex":0.0003861928,"about_ca_system_score_gemma":0.0006335105,"threshold_uncertainty_score":0.044669807},"labels":[],"label_agreement":null},{"id":"W3101574558","doi":"10.1186/s13040-016-0108-2","title":"ProtNN: fast and accurate protein 3D-structure classification in structural and topological space","year":2016,"lang":"en","type":"article","venue":"BioData Mining","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Space (punctuation); Topology (electrical circuits); Topological space; Data mining; Pattern recognition (psychology); Artificial intelligence; Mathematics; Pure mathematics; Combinatorics","score_opus":0.016169668764608368,"score_gpt":0.2731470377325991,"score_spread":0.2569773689679907,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3101574558","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.096809186,0.0032674556,0.7183175,0.0007931475,0.00049978314,0.0005699092,0.027450982,0.14600994,0.0062821694],"genre_scores_gemma":[0.19304045,0.0009587271,0.73236495,0.0002788589,0.0001005591,0.00054924795,0.0640672,0.0020806394,0.0065593626],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99887234,0.000101399266,0.00007664303,0.00034248078,0.00051954144,0.00008751272],"domain_scores_gemma":[0.9988887,0.00021862634,0.00017731127,0.00030089665,0.00033528765,0.00007920448],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00080039393,0.0017393543,0.0014584194,0.0031058355,0.0009552904,0.0018315582,0.0024064092,0.0013928766,0.0053275567],"category_scores_gemma":[0.002512963,0.0005810703,0.0012838709,0.0024374805,0.00043423442,0.002169662,0.002042857,0.00097044284,0.0053793527],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000797521,0.00029457046,0.012099551,0.0011398908,0.00025011552,0.0003067714,0.00018461386,0.035559364,0.034476068,0.0044625886,0.1338115,0.77661747],"study_design_scores_gemma":[0.00008267788,0.0001476598,0.0045518177,0.000057924477,0.000048972553,0.00059325405,0.0001286824,0.942075,0.019519724,0.0066283625,0.026105061,0.000060875755],"about_ca_topic_score_codex":0.006282649,"about_ca_topic_score_gemma":0.01044758,"teacher_disagreement_score":0.006282649,"about_ca_system_score_codex":0.0011373834,"about_ca_system_score_gemma":0.0011370443,"threshold_uncertainty_score":0.017822444},"labels":[],"label_agreement":null},{"id":"W3126232929","doi":"10.1186/s13040-021-00244-z","title":"The Matthews correlation coefficient (MCC) is more reliable than balanced accuracy, bookmaker informedness, and markedness in two-class confusion matrix evaluation","year":2021,"lang":"en","type":"article","venue":"BioData Mining","topic":"Imbalanced Data Classification Techniques","field":"Computer Science","cited_by":893,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Krembil Foundation","funders":"Deutsche Forschungsgemeinschaft","keywords":"Confusion matrix; Correlation; False positive paradox; Computer science; Confusion; False positives and false negatives; Classifier (UML); Statistics; Artificial intelligence; Metric (unit); Correlation coefficient; Machine learning; Data mining; Mathematics; Psychology; Operations management","score_opus":0.029274177312831263,"score_gpt":0.33077275706615017,"score_spread":0.3014985797533189,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3126232929","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.63035774,0.007464869,0.3337457,0.0022520798,0.0009788257,0.0005293683,0.0035446393,0.0043198736,0.016806932],"genre_scores_gemma":[0.9578433,0.00029579006,0.038165092,0.00024431964,0.00025636095,0.00015377674,0.0015162769,0.00023166137,0.0012933959],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98146665,0.005495087,0.0025518108,0.003180954,0.0066273003,0.0006782635],"domain_scores_gemma":[0.8398484,0.10258917,0.02223467,0.008470453,0.023687406,0.003169879],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01584418,0.0015452036,0.0015110709,0.007663949,0.0013663475,0.0032967268,0.0010549694,0.0019153301,0.0018176283],"category_scores_gemma":[0.11076356,0.00033500406,0.0008541324,0.0043991352,0.0023043193,0.002821813,0.0018214865,0.0014286769,0.0009953736],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024923997,0.00036101727,0.42262766,0.0018222617,0.0026262098,0.00093730545,0.0021934258,0.13442937,0.018697899,0.020015959,0.043514688,0.35028175],"study_design_scores_gemma":[0.000099068806,0.0015876819,0.17999624,0.0005762462,0.000605116,0.0015482736,0.0011539743,0.7163308,0.041929428,0.040945873,0.0146318115,0.0005954534],"about_ca_topic_score_codex":0.0038179145,"about_ca_topic_score_gemma":0.0050589275,"teacher_disagreement_score":0.01584418,"about_ca_system_score_codex":0.0014003182,"about_ca_system_score_gemma":0.0013194487,"threshold_uncertainty_score":0.083793044},"labels":[],"label_agreement":null},{"id":"W3128825412","doi":"10.1186/s13040-021-00235-0","title":"Data analytics and clinical feature ranking of medical records of patients with sepsis","year":2021,"lang":"en","type":"article","venue":"BioData Mining","topic":"Sepsis Diagnosis and Treatment","field":"Medicine","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Krembil Foundation","funders":"University of Toronto","keywords":"Sepsis; Medical record; Septic shock; Medicine; Context (archaeology); Ranking (information retrieval); Electronic medical record; Machine learning; Binary classification; Computer science; Logistic regression; SOFA score; Artificial intelligence; Intensive care medicine; Data mining; Emergency medicine; Internal medicine; Support vector machine","score_opus":0.21828630754067863,"score_gpt":0.41707643113932913,"score_spread":0.1987901235986505,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3128825412","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94152457,0.0025002458,0.01636924,0.0022378995,0.00019830407,0.00028621985,0.03477471,0.0007163283,0.0013924964],"genre_scores_gemma":[0.9510968,0.00033927453,0.0166043,0.00013298466,0.00011625972,0.00013428828,0.031380743,0.00001306634,0.00018230421],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9952058,0.0016340121,0.00083299814,0.00081718573,0.0011812145,0.00032884645],"domain_scores_gemma":[0.97869724,0.012583347,0.0036206304,0.0015400201,0.0030197022,0.00053913286],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003860968,0.0007202197,0.00073977874,0.0068945466,0.00047155676,0.0015387909,0.0007600565,0.00083233,0.00074295944],"category_scores_gemma":[0.024153637,0.0001682403,0.0009807163,0.006003263,0.00036107798,0.0012212185,0.00075781095,0.0008950296,0.00046949068],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008040715,0.000716246,0.83591205,0.0005417802,0.00042049072,0.00042362476,0.00026798638,0.021731025,0.0020421508,0.00080126256,0.007942503,0.1283969],"study_design_scores_gemma":[0.000083557985,0.00073743076,0.6946102,0.00025207506,0.00022608234,0.001045475,0.0012677965,0.28476575,0.0056029693,0.003548408,0.007770318,0.00008991408],"about_ca_topic_score_codex":0.0060699917,"about_ca_topic_score_gemma":0.0060515334,"teacher_disagreement_score":0.0068945466,"about_ca_system_score_codex":0.0012005052,"about_ca_system_score_gemma":0.0013443519,"threshold_uncertainty_score":0.020419002},"labels":[],"label_agreement":null},{"id":"W3136786679","doi":"10.1186/s13040-021-00251-0","title":"Indels in SARS-CoV-2 occur at template-switching hotspots","year":2021,"lang":"en","type":"article","venue":"BioData Mining","topic":"SARS-CoV-2 and COVID-19 Research","field":"Medicine","cited_by":39,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"National Science Foundation of Sri Lanka; Stanford Bio-X; National Science Foundation","keywords":"Indel; Genetics; Biology; RNA; Computational biology; Homologous recombination; RNA polymerase; Recombination; Evolutionary biology; Gene; Single-nucleotide polymorphism; Genotype","score_opus":0.12193423570309297,"score_gpt":0.3875826429091648,"score_spread":0.2656484072060718,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3136786679","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9894856,0.0018486911,0.00308277,0.00005148704,0.0000333798,0.000033591143,0.0035119369,0.00023552476,0.0017169395],"genre_scores_gemma":[0.98452824,0.0004100147,0.006682403,0.00014179542,0.000039287435,0.00003387974,0.0073226388,0.000096145646,0.00074554415],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993487,0.000055634955,0.00006551887,0.00027357522,0.00018815232,0.00006845828],"domain_scores_gemma":[0.99847835,0.00045710828,0.0006120832,0.00015780519,0.00016928183,0.00012541888],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00038194546,0.00033966897,0.00054820377,0.0017756172,0.00046663714,0.0005669673,0.00030447944,0.00062849163,0.0022466262],"category_scores_gemma":[0.0013622683,0.0002859429,0.0007087725,0.001963735,0.00030387565,0.0003504082,0.00037408524,0.0005563497,0.0009365473],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017163785,0.00015042283,0.19555345,0.0010158495,0.00048179485,0.004501267,0.0010430187,0.0025149593,0.7582355,0.000883136,0.0011560926,0.032748215],"study_design_scores_gemma":[0.000051872903,0.00027421667,0.90923536,0.00010362838,0.0002816729,0.006073048,0.00041791605,0.0033177785,0.06901236,0.0010840103,0.010058079,0.00009008853],"about_ca_topic_score_codex":0.0007382452,"about_ca_topic_score_gemma":0.0020960765,"teacher_disagreement_score":0.0022466262,"about_ca_system_score_codex":0.0002076537,"about_ca_system_score_gemma":0.00018993986,"threshold_uncertainty_score":0.007515669},"labels":[],"label_agreement":null},{"id":"W3141242658","doi":"10.1186/s13040-021-00249-8","title":"Prescreening and treatment of aortic dissection through an analysis of infinite-dimension data","year":2021,"lang":"en","type":"article","venue":"BioData Mining","topic":"Aortic Disease and Treatment Approaches","field":"Medicine","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Prince Edward Island; University of Waterloo","funders":"Shanghai Jiao Tong University; Shanghai Municipal Health Bureau; National Natural Science Foundation of China; School of Medicine, Shanghai Jiao Tong University; Comanche Nation","keywords":"Aortic dissection; Medicine; Cardiology; Blood pressure; Internal medicine; Mortality rate; Disease; Incidence (geometry); Medical diagnosis; Psychological intervention; Dissection (medical); Intensive care medicine; Emergency medicine; Surgery; Radiology; Aorta","score_opus":0.15880538879975056,"score_gpt":0.3612328588707938,"score_spread":0.20242747007104325,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3141242658","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7928641,0.00080741185,0.20200588,0.0007877876,0.000061072584,0.000119871176,0.0015709355,0.00061811815,0.0011649074],"genre_scores_gemma":[0.95834935,0.00021165606,0.039550405,0.000045183548,0.000032427568,0.00005138016,0.001432987,0.000015297377,0.00031131576],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9994222,0.00016980227,0.00007773736,0.00014816887,0.00011887347,0.00006324338],"domain_scores_gemma":[0.99632543,0.002378467,0.0005540192,0.00025294095,0.00036307902,0.00012602132],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019441637,0.00062154443,0.00059364847,0.0027729361,0.00033678758,0.00092570327,0.00059733563,0.000564115,0.0007886972],"category_scores_gemma":[0.0060863136,0.00017576134,0.00072901166,0.0008982123,0.000358063,0.0006720597,0.00052348274,0.00055404246,0.00023015974],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007798107,0.0007633328,0.46571338,0.00023343676,0.00028824437,0.00083141105,0.00028960634,0.26948094,0.0065515954,0.0017854242,0.0023476526,0.25093508],"study_design_scores_gemma":[0.0000098296305,0.00017387104,0.057569727,0.000032747135,0.000040191673,0.00020820317,0.00008705495,0.93788713,0.0015788334,0.0017741028,0.0006136315,0.000024720011],"about_ca_topic_score_codex":0.0031489981,"about_ca_topic_score_gemma":0.0034909358,"teacher_disagreement_score":0.0031489981,"about_ca_system_score_codex":0.0005216574,"about_ca_system_score_gemma":0.0007221454,"threshold_uncertainty_score":0.010281861},"labels":[],"label_agreement":null},{"id":"W3215710945","doi":"10.1186/s13040-021-00281-8","title":"Development of glaucoma predictive model and risk factors assessment based on supervised models","year":2021,"lang":"en","type":"article","venue":"BioData Mining","topic":"Retinal Imaging and Analysis","field":"Medicine","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Glaucoma; Machine learning; Artificial intelligence; Data science; Risk analysis (engineering); Medicine; Ophthalmology","score_opus":0.055093053541754644,"score_gpt":0.3059507429795755,"score_spread":0.25085768943782083,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3215710945","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11718565,0.00093635573,0.8764417,0.00063007214,0.00011713864,0.00021998465,0.0006652683,0.0010720986,0.0027317298],"genre_scores_gemma":[0.86767626,0.0007224717,0.12689435,0.00017827004,0.0001744233,0.00040174078,0.0012584656,0.000043786553,0.002650176],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9994029,0.00016210615,0.00004853719,0.0001647108,0.00015796424,0.000063885935],"domain_scores_gemma":[0.9988939,0.00046460028,0.00011883284,0.00006447873,0.00041510808,0.000043087603],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014265398,0.00071834016,0.0008330918,0.0013046222,0.00035372627,0.000787708,0.0011425535,0.0006513964,0.0011778303],"category_scores_gemma":[0.0032900618,0.00030711672,0.0011295144,0.000569972,0.00021513375,0.00081911834,0.0005796239,0.0009260851,0.00038303097],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00021174157,0.00040358846,0.050351262,0.00018329722,0.00035182646,0.00035223635,0.00017544972,0.66253555,0.0029818334,0.0050853346,0.005277843,0.27209005],"study_design_scores_gemma":[0.0000048854727,0.000032225613,0.0017960015,0.000015062701,0.00002578553,0.000036975962,0.000013331788,0.99610114,0.00030553943,0.0013132266,0.00034920443,0.000006667652],"about_ca_topic_score_codex":0.007316375,"about_ca_topic_score_gemma":0.005520941,"teacher_disagreement_score":0.007316375,"about_ca_system_score_codex":0.0005192049,"about_ca_system_score_gemma":0.0011834978,"threshold_uncertainty_score":0.014547586},"labels":[],"label_agreement":null},{"id":"W4234031396","doi":"10.1186/preaccept-8590947901180703","title":"Integrative genomics and transcriptomics analysis of human embryonic and induced pluripotent stem cells","year":2014,"lang":"en","type":"article","venue":"BioData Mining","topic":"Genomic variations and chromosomal abnormalities","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Lunenfeld-Tanenbaum Research Institute","funders":"Emil Aaltosen Säätiö; European Commission","keywords":"Biology; Induced pluripotent stem cell; Genetics; Single-nucleotide polymorphism; Gene; SNP array; Copy-number variation; Genomics; DNA microarray; Transcriptome; Embryonic stem cell; Human genome; Functional genomics; Phenotype; Exon; Genome; Computational biology; Gene expression; Genotype","score_opus":0.017187619078278094,"score_gpt":0.23010869046941393,"score_spread":0.21292107139113584,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4234031396","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.88554627,0.0032428605,0.09178872,0.00013922976,0.00003339045,0.00011389955,0.015435812,0.0006173403,0.003082542],"genre_scores_gemma":[0.8597494,0.0020316173,0.093103975,0.00015184241,0.000026185435,0.00022210438,0.042955685,0.00011264648,0.001646676],"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9997837,0.00003058881,0.000017582635,0.00007586544,0.000071951705,0.000020237894],"domain_scores_gemma":[0.99992716,0.000026368834,0.000015285828,0.000009205098,0.00001573947,0.000006223392],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00027650804,0.00025247567,0.00040608525,0.00067378423,0.00015788169,0.00030843948,0.00018517542,0.00012924032,0.00057533017],"category_scores_gemma":[0.0002694049,0.00010475114,0.00055563456,0.0008042315,0.00012414448,0.00012209796,0.00020953678,0.00016601852,0.00020408051],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001680615,0.000023757711,0.005089585,0.00019354498,0.00006830358,0.00020874939,0.0000691992,0.0025044922,0.9690971,0.0010364851,0.00020393997,0.021336883],"study_design_scores_gemma":[0.000056399298,0.00048518938,0.2355541,0.00005719373,0.00046702722,0.0014779338,0.00025325155,0.0482942,0.68432957,0.0047098584,0.02426863,0.00004664203],"about_ca_topic_score_codex":0.0006936031,"about_ca_topic_score_gemma":0.0008670308,"teacher_disagreement_score":0.0006936031,"about_ca_system_score_codex":0.00021307459,"about_ca_system_score_gemma":0.00034323405,"threshold_uncertainty_score":0.0019246936},"labels":[],"label_agreement":null},{"id":"W4307995431","doi":"10.1186/s13040-022-00312-y","title":"Towards a potential pan-cancer prognostic signature for gene expression based on probesets and ensemble machine learning","year":2022,"lang":"en","type":"article","venue":"BioData Mining","topic":"Bioinformatics and Genomic Networks","field":"Biochemistry, Genetics and Molecular Biology","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Krembil Foundation; University of Toronto","funders":"","keywords":"Signature (topology); Gene signature; Cancer; Gene; Computational biology; Random forest; Relevance (law); Microarray analysis techniques; Biology; Microarray; Bioinformatics; Computer science; Gene expression; Machine learning; Genetics; Mathematics","score_opus":0.012604550967461287,"score_gpt":0.24221652434108,"score_spread":0.2296119733736187,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4307995431","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24938068,0.0019074264,0.74151486,0.0005200373,0.00012985572,0.00015500783,0.0022893273,0.0026263094,0.0014764103],"genre_scores_gemma":[0.80655366,0.00055553456,0.18632805,0.0002379022,0.00010745409,0.00023031613,0.0049767103,0.00008935136,0.0009209712],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99909055,0.00025513393,0.000071466755,0.00025259866,0.00020890379,0.00012131148],"domain_scores_gemma":[0.9990565,0.00037010928,0.00015193004,0.00011994463,0.0002443864,0.000057127094],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015089688,0.0009072078,0.0011709572,0.0025749356,0.0003105898,0.0007368958,0.00072024646,0.00066073204,0.00077341223],"category_scores_gemma":[0.0029841864,0.0002054372,0.00091795705,0.0023968003,0.0003349185,0.0009391695,0.0007079436,0.00070552155,0.00036494352],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012434745,0.00033918762,0.08429143,0.0004684147,0.0005891087,0.00061279174,0.00020516146,0.18449524,0.088093296,0.0047691716,0.005282061,0.6296106],"study_design_scores_gemma":[0.000042643067,0.00034384034,0.015993966,0.000038256214,0.00022204572,0.00035496583,0.00006464587,0.94602203,0.023355402,0.009881041,0.003629995,0.000051120805],"about_ca_topic_score_codex":0.0013638722,"about_ca_topic_score_gemma":0.0015757424,"teacher_disagreement_score":0.0025749356,"about_ca_system_score_codex":0.00039761848,"about_ca_system_score_gemma":0.00060566864,"threshold_uncertainty_score":0.007980287},"labels":[],"label_agreement":null},{"id":"W4321214126","doi":"10.1186/s13040-023-00322-4","title":"The Matthews correlation coefficient (MCC) should replace the ROC AUC as the standard metric for assessing binary classification","year":2023,"lang":"en","type":"article","venue":"BioData Mining","topic":"Imbalanced Data Classification Techniques","field":"Computer Science","cited_by":513,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Receiver operating characteristic; Matthews correlation coefficient; Binary classification; Statistics; Confusion matrix; False positive rate; Artificial intelligence; Correlation; Mathematics; Binary number; Sensitivity (control systems); Metric (unit); Classifier (UML); Computer science; Pattern recognition (psychology); Machine learning; Support vector machine; Arithmetic","score_opus":0.10998735913183141,"score_gpt":0.36279671487678333,"score_spread":0.2528093557449519,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4321214126","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028617183,0.18328878,0.5222214,0.10144067,0.07616145,0.0012663502,0.010737325,0.016711572,0.059555277],"genre_scores_gemma":[0.33909854,0.061279092,0.45696115,0.05931128,0.026797712,0.0033729337,0.01172351,0.0058960533,0.035559736],"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.981873,0.005489669,0.0023558093,0.0025918228,0.007197795,0.0004919061],"domain_scores_gemma":[0.93822765,0.034521397,0.0071689817,0.005280702,0.013355837,0.0014454586],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.015305564,0.0027569144,0.004081455,0.0039412435,0.0009771422,0.004764434,0.003884683,0.005842996,0.0068738614],"category_scores_gemma":[0.098277435,0.0008002041,0.001730083,0.006211227,0.003612601,0.0047042225,0.0017254985,0.0059769438,0.012511089],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00054439536,0.00012626736,0.017333936,0.003481501,0.0008541755,0.00071329693,0.0002993405,0.0043026814,0.0045021754,0.050688233,0.56764483,0.34950918],"study_design_scores_gemma":[0.00021524848,0.0011370275,0.041185588,0.0033058065,0.0009035009,0.0062145204,0.0005875552,0.04324102,0.026488729,0.18044636,0.6952102,0.0010644055],"about_ca_topic_score_codex":0.003513164,"about_ca_topic_score_gemma":0.0033432979,"teacher_disagreement_score":0.9846944,"about_ca_system_score_codex":0.0028636926,"about_ca_system_score_gemma":0.003009216,"threshold_uncertainty_score":0.08094454},"labels":[],"label_agreement":null},{"id":"W4321598956","doi":"10.1186/s13040-023-00326-0","title":"Ten simple rules for providing bioinformatics support within a hospital","year":2023,"lang":"en","type":"article","venue":"BioData Mining","topic":"Genetics, Bioinformatics, and Biomedical Research","field":"Biochemistry, Genetics and Molecular Biology","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Simple (philosophy); Computer science; Data science; Data mining; Bioinformatics; Biology","score_opus":0.037680212675283105,"score_gpt":0.3102693955737785,"score_spread":0.2725891828984954,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4321598956","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.028752884,0.0007073494,0.84324294,0.046212573,0.0019365825,0.0030466458,0.0014485781,0.008860725,0.06579158],"genre_scores_gemma":[0.07127856,0.00039865848,0.915807,0.0036270153,0.00019432946,0.000825625,0.00080231536,0.00031682008,0.0067496784],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.97478,0.009957201,0.004974379,0.0019855245,0.006619441,0.0016833781],"domain_scores_gemma":[0.93300325,0.034616128,0.006068912,0.0066449116,0.014965068,0.004701682],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020986306,0.0013521648,0.0009724489,0.003554346,0.0038605586,0.008221486,0.0044313893,0.0068145082,0.009414572],"category_scores_gemma":[0.06739782,0.0012785371,0.0015305555,0.0025917424,0.004897421,0.009127513,0.005298252,0.005162581,0.0061276467],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005951379,0.0011939064,0.026339509,0.0015930408,0.00020960675,0.0026372655,0.0049910285,0.02750521,0.0045214416,0.4652186,0.16389033,0.3013049],"study_design_scores_gemma":[0.00059005676,0.0006044069,0.0066742143,0.00222286,0.00038474132,0.0019016367,0.0050223707,0.09509418,0.009023091,0.479217,0.39860266,0.0006627478],"about_ca_topic_score_codex":0.010928732,"about_ca_topic_score_gemma":0.016214369,"teacher_disagreement_score":0.020986306,"about_ca_system_score_codex":0.002220355,"about_ca_system_score_gemma":0.008702688,"threshold_uncertainty_score":0.110987484},"labels":[],"label_agreement":null},{"id":"W4323050290","doi":"10.1186/s13040-023-00325-1","title":"Signature literature review reveals AHCY, DPYSL3, and NME1 as the most recurrent prognostic genes for neuroblastoma","year":2023,"lang":"en","type":"article","venue":"BioData Mining","topic":"Neuroblastoma Research and Treatments","field":"Medicine","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Neuroblastoma; Gene; Bioinformatics; Oncology; Biology; Computational biology; Medicine; Internal medicine; Genetics","score_opus":0.04151356111928294,"score_gpt":0.3423840202260795,"score_spread":0.30087045910679655,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4323050290","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.012329048,0.97792953,0.0008149726,0.0027566974,0.00056556525,0.00003887575,0.0031878727,0.000046810852,0.0023306897],"genre_scores_gemma":[0.0613768,0.92890394,0.0017295034,0.0019000043,0.00050763483,0.00005867295,0.0046103816,0.000023034383,0.00089005765],"study_design_codex":"design_other","study_design_gemma":"systematic_review","domain_scores_codex":[0.99910957,0.00014243527,0.00031135094,0.000159896,0.00022465794,0.000052112955],"domain_scores_gemma":[0.9957755,0.002509169,0.00059344806,0.00013566083,0.0008736047,0.00011271803],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014440846,0.0003784365,0.0011291563,0.0074142246,0.00042968438,0.0010481232,0.000642575,0.0006515063,0.0032179093],"category_scores_gemma":[0.005771402,0.00015139261,0.0014927399,0.008475335,0.00047240034,0.000810843,0.00046009952,0.0005363362,0.00069626485],"study_design_candidate":"systematic_review","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011115464,0.000084698346,0.037270013,0.14470164,0.0030562559,0.0017454647,0.0005744621,0.00068629975,0.0058264057,0.002184697,0.052630942,0.7501276],"study_design_scores_gemma":[0.00012345203,0.00044163497,0.12873961,0.10800755,0.027133172,0.008565511,0.0012212663,0.0008372271,0.007433284,0.0036936125,0.7136644,0.00013923744],"about_ca_topic_score_codex":0.0029774748,"about_ca_topic_score_gemma":0.0064969403,"teacher_disagreement_score":0.0074142246,"about_ca_system_score_codex":0.0008394192,"about_ca_system_score_gemma":0.0033466916,"threshold_uncertainty_score":0.0107649565},"labels":[],"label_agreement":null},{"id":"W4387402455","doi":"10.1186/s13040-023-00343-z","title":"Quantum analysis of squiggle data","year":2023,"lang":"en","type":"article","venue":"BioData Mining","topic":"Quantum Computing Algorithms and Architecture","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"Alberta Innovates; Ministero dello Sviluppo Economico; Government of Alberta; Genome Canada","keywords":"Nanopore sequencing; Computer science; Curse of dimensionality; Algorithm; Quantum; IBM; Quantum computer; Current (fluid); Computer engineering; DNA sequencing; Theoretical computer science; Artificial intelligence; DNA; Electrical engineering; Physics","score_opus":0.07099168818858985,"score_gpt":0.30829258730085923,"score_spread":0.23730089911226937,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387402455","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14325851,0.00019601967,0.8509939,0.00047430384,0.000053813266,0.00004879441,0.00024133213,0.0012567551,0.0034766314],"genre_scores_gemma":[0.8000981,0.00020130312,0.19600421,0.00014095503,0.000041368836,0.000047807116,0.00044351228,0.00018056556,0.002842104],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99948394,0.000091988986,0.000026452355,0.000090173635,0.00024269568,0.00006473295],"domain_scores_gemma":[0.9984553,0.00065057795,0.00010835398,0.00030220367,0.00042436385,0.000059230122],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007173202,0.00027816353,0.00045708995,0.0011305679,0.0005121778,0.0008885052,0.0006435525,0.00038092188,0.0036267582],"category_scores_gemma":[0.0039467863,0.00016402804,0.00041688618,0.0010452982,0.0007447557,0.0017339563,0.000774061,0.0007419756,0.0005091871],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00053172855,0.00021747616,0.0072880764,0.0002745416,0.00013506647,0.00070812274,0.00055483077,0.2701764,0.15849581,0.2278606,0.006847782,0.32690957],"study_design_scores_gemma":[0.0000069677258,0.000028147446,0.0014535567,0.0000064251617,0.000006136267,0.00007438864,0.00007055237,0.93841845,0.016226126,0.041870646,0.0018233737,0.0000152300345],"about_ca_topic_score_codex":0.0017011337,"about_ca_topic_score_gemma":0.0015738859,"teacher_disagreement_score":0.0036267582,"about_ca_system_score_codex":0.00061398256,"about_ca_system_score_gemma":0.0006188573,"threshold_uncertainty_score":0.012132704},"labels":[],"label_agreement":null},{"id":"W4387813874","doi":"10.1186/s13040-023-00344-y","title":"Prescription pattern analysis of Type 2 Diabetes Mellitus: a cross-sectional study in Isfahan, Iran","year":2023,"lang":"en","type":"article","venue":"BioData Mining","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada); University of Toronto","funders":"","keywords":"Medical prescription; Medicine; Diabetes mellitus; Polypharmacy; Type 2 Diabetes Mellitus; Cross-sectional study; Type 2 diabetes; Internal medicine; Pharmacology; Endocrinology","score_opus":0.08700440662326508,"score_gpt":0.3629943011081319,"score_spread":0.2759898944848668,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387813874","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9994803,0.00010677609,0.00004511747,0.00003153837,0.0000034242132,0.000015531436,0.0002073461,0.0000010777694,0.00010885922],"genre_scores_gemma":[0.999079,0.00018676787,0.00023912545,0.000054357784,0.000010698886,0.000027876154,0.00032879593,0.0000012877541,0.00007213815],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9993048,0.00016335251,0.000103614766,0.00014008733,0.00020682627,0.00008137067],"domain_scores_gemma":[0.99885905,0.00026951035,0.0004915571,0.000071722316,0.00018976552,0.00011843317],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011678165,0.00023768037,0.00043688962,0.0012480904,0.000500582,0.00045397304,0.00037949582,0.00037311422,0.00090837973],"category_scores_gemma":[0.0016332663,0.0003217431,0.0005929665,0.0019308645,0.00022430599,0.000430513,0.00030380354,0.0004735116,0.00012753898],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017245784,0.000053816984,0.9989452,0.000011045134,0.000033215336,0.000043852233,0.000099520235,0.00001156216,0.00004394073,0.000004665636,0.000042225365,0.0006936913],"study_design_scores_gemma":[0.0000048718816,0.00008764526,0.9989617,0.000005800657,0.000025589745,0.00023402914,0.00042401135,0.00012521113,0.000028553548,0.0000060976454,0.00009395658,0.000002409223],"about_ca_topic_score_codex":0.00891997,"about_ca_topic_score_gemma":0.011451813,"teacher_disagreement_score":0.00891997,"about_ca_system_score_codex":0.00047948115,"about_ca_system_score_gemma":0.00068358023,"threshold_uncertainty_score":0.017736137},"labels":[],"label_agreement":null},{"id":"W4390605262","doi":"10.1186/s13040-023-00352-y","title":"Machine learning approaches to identify systemic lupus erythematosus in anti-nuclear antibody-positive patients using genomic data and electronic health records","year":2024,"lang":"en","type":"article","venue":"BioData Mining","topic":"Systemic Lupus Erythematosus Research","field":"Medicine","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Academia Sinica; Taichung Veterans General Hospital; National Science and Technology Council","keywords":"Logistic regression; Single-nucleotide polymorphism; Medicine; Random forest; Titer; Gradient boosting; Artificial intelligence; Medical record; Machine learning; Systemic lupus erythematosus; Internal medicine; Cohort; Support vector machine; Oncology; Immunology; Antibody; Computer science; Disease; Biology; Genetics; Genotype","score_opus":0.11259950949713039,"score_gpt":0.36030285822970487,"score_spread":0.2477033487325745,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390605262","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.66764,0.007725693,0.30356842,0.0063689053,0.00023924839,0.00072033866,0.0077146455,0.001228066,0.0047947713],"genre_scores_gemma":[0.91640705,0.0009356759,0.07857992,0.00039216332,0.00017582506,0.00024478522,0.0029078932,0.000013010881,0.0003436593],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9969759,0.0019530479,0.00024268098,0.00042487134,0.000284428,0.00011906944],"domain_scores_gemma":[0.99176484,0.005722207,0.0011717279,0.00042444398,0.00074224436,0.00017456031],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005758174,0.00069985085,0.00093832245,0.0065076076,0.00032634492,0.0012537566,0.0009745558,0.0008330497,0.0009517414],"category_scores_gemma":[0.013911965,0.00025929342,0.0008764535,0.0029154574,0.0002405995,0.00077651977,0.00073860167,0.0010064768,0.00043127884],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037144058,0.0008261614,0.689024,0.00034298428,0.0010435137,0.00028421814,0.00019741307,0.038406923,0.0011706608,0.0012923641,0.0035091355,0.2635312],"study_design_scores_gemma":[0.000093454706,0.00031929777,0.13502511,0.00025697457,0.00030334297,0.0003559827,0.0003855766,0.8503098,0.0012902895,0.009468509,0.002145876,0.000045845747],"about_ca_topic_score_codex":0.00280392,"about_ca_topic_score_gemma":0.0033620608,"teacher_disagreement_score":0.0065076076,"about_ca_system_score_codex":0.0006092443,"about_ca_system_score_gemma":0.0007938407,"threshold_uncertainty_score":0.03045255},"labels":[],"label_agreement":null},{"id":"W4391353343","doi":"10.1186/s13040-024-00355-3","title":"Revealing third-order interactions through the integration of machine learning and entropy methods in genomic studies","year":2024,"lang":"en","type":"article","venue":"BioData Mining","topic":"Genetic Associations and Epidemiology","field":"Biochemistry, Genetics and Molecular Biology","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Institute on Aging; Canadian Institutes of Health Research; Genentech; IXICO; H. Lundbeck A/S; Servier; National Institutes of Health; Türkiye Bilimsel ve Teknolojik Araştırma Kurumu; Eisai; Pfizer; Novartis Pharmaceuticals Corporation; U.S. Department of Defense; Meso Scale Diagnostics; Northern California Institute for Research and Education; F. Hoffmann-La Roche; University of Southern California; BioClinica; Bristol-Myers Squibb; Eli Lilly and Company; Biogen","keywords":"Genome-wide association study; Computer science; Epistasis; Single-nucleotide polymorphism; Genetic association; SNP; Computational biology; Artificial intelligence; Data mining; Machine learning; Biology; Genetics; Genotype; Gene","score_opus":0.07409299352930675,"score_gpt":0.42225888025484476,"score_spread":0.348165886725538,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391353343","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03914229,0.00046383357,0.9583568,0.0001868154,0.00002398898,0.000056519373,0.0003275497,0.00079083804,0.0006514255],"genre_scores_gemma":[0.500574,0.00057072117,0.4959712,0.00016179973,0.00008948391,0.0003221109,0.0014058711,0.00021667617,0.0006880604],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9979101,0.0011229379,0.00011361242,0.00035819475,0.00039352267,0.00010160181],"domain_scores_gemma":[0.99223447,0.0062317178,0.000507935,0.000528869,0.00034733093,0.00014974376],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047519878,0.0009995464,0.0009977149,0.0031994663,0.0005365888,0.0014163123,0.00093493407,0.0005932856,0.0012259164],"category_scores_gemma":[0.009189226,0.00035157314,0.0026251026,0.0018469355,0.0007328783,0.0010857447,0.0015656248,0.0011728839,0.00031599402],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00050591514,0.00037074322,0.07380475,0.00066060177,0.0013650977,0.0010421859,0.00063482515,0.5083421,0.02719044,0.042825866,0.0024177881,0.34083974],"study_design_scores_gemma":[0.000015976295,0.00006202018,0.009322106,0.000026400136,0.00010369997,0.00012618318,0.00003865602,0.9427505,0.0035325151,0.042557217,0.0014158556,0.000048847105],"about_ca_topic_score_codex":0.002529044,"about_ca_topic_score_gemma":0.0028659578,"teacher_disagreement_score":0.0047519878,"about_ca_system_score_codex":0.0005602583,"about_ca_system_score_gemma":0.0011456563,"threshold_uncertainty_score":0.025131226},"labels":[],"label_agreement":null},{"id":"W4402164248","doi":"10.1186/s13040-024-00380-2","title":"Seven quick tips for gene-focused computational pangenomic analysis","year":2024,"lang":"en","type":"article","venue":"BioData Mining","topic":"Cell Image Analysis Techniques","field":"Biochemistry, Genetics and Molecular Biology","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Università degli Studi di Parma; Università degli Studi di Trento","keywords":"Computer science; Data science; Computational biology; Data mining; Biology","score_opus":0.021258712347645495,"score_gpt":0.29616306982242957,"score_spread":0.2749043574747841,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402164248","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0018729554,0.0065323957,0.90897524,0.020721687,0.008036988,0.00042748198,0.0016618614,0.04521663,0.006554796],"genre_scores_gemma":[0.005907183,0.0035418682,0.963959,0.0053573116,0.0020858448,0.00050947967,0.001738397,0.010976749,0.0059241727],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99159086,0.0028749611,0.00094659714,0.00074618193,0.0033935853,0.00044772885],"domain_scores_gemma":[0.93957067,0.031680007,0.0018393064,0.0070461375,0.017260304,0.002603615],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009843729,0.004308954,0.0020113518,0.0052840654,0.0019807552,0.0069778604,0.00599969,0.005123996,0.043898802],"category_scores_gemma":[0.09296288,0.0028686735,0.0024204405,0.004230991,0.0025854954,0.011732458,0.0056228777,0.014927741,0.049700968],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004723371,0.00023770654,0.0015922652,0.0019336218,0.00019475434,0.0012373892,0.00070831744,0.004087774,0.009503412,0.023643076,0.5675443,0.3888451],"study_design_scores_gemma":[0.00029712592,0.00021605779,0.0015329085,0.0017578611,0.00015076072,0.002030245,0.0008026191,0.024627697,0.019556236,0.1883968,0.7601055,0.0005261191],"about_ca_topic_score_codex":0.001144494,"about_ca_topic_score_gemma":0.0022745263,"teacher_disagreement_score":0.043898802,"about_ca_system_score_codex":0.00133926,"about_ca_system_score_gemma":0.0023445452,"threshold_uncertainty_score":0.14685607},"labels":[],"label_agreement":null},{"id":"W4404300486","doi":"10.1186/s13040-024-00400-1","title":"Deciphering the tissue-specific functional effect of Alzheimer risk SNPs with deep genome annotation","year":2024,"lang":"en","type":"article","venue":"BioData Mining","topic":"Bioinformatics and Genomic Networks","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"National Institutes of Health; Genentech; IXICO; H. Lundbeck A/S; Servier; Eisai; Northern California Institute for Research and Education; University of Southern California; Biogen; BioClinica; Meso Scale Diagnostics; U.S. Department of Defense; Alzheimer's Disease Neuroimaging Initiative; Novartis Pharmaceuticals Corporation; Pfizer; Eli Lilly and Company; Bristol-Myers Squibb; National Institute on Aging; Alzheimer's Association; Canadian Institutes of Health Research; National Science Foundation","keywords":"Annotation; Genome-wide association study; Single-nucleotide polymorphism; Genome; Computer science; Computational biology; Biology; Data science; Artificial intelligence; Genetics; Gene; Genotype","score_opus":0.013763408373050517,"score_gpt":0.23232748207770818,"score_spread":0.21856407370465766,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404300486","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.85955113,0.0022130047,0.1290246,0.00060377753,0.000053546974,0.000026561474,0.0055468264,0.00086153403,0.0021190555],"genre_scores_gemma":[0.9606225,0.0006100857,0.03338497,0.00017374726,0.000014286235,0.000032102886,0.004419969,0.000094384966,0.0006479027],"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9997346,0.00006830976,0.000013337892,0.00008999124,0.000044549874,0.000049183618],"domain_scores_gemma":[0.9995427,0.0002577226,0.000059500137,0.00005971308,0.000041555537,0.000038825703],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00072126754,0.00040547113,0.0005608842,0.0008316546,0.00031016246,0.0006139761,0.00045397822,0.00035637917,0.0010897861],"category_scores_gemma":[0.0015752187,0.00020768064,0.0005360449,0.00082733424,0.00024923537,0.0004764232,0.00069839845,0.000619831,0.00024635918],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010323928,0.00028728013,0.26626298,0.00097015226,0.0010327966,0.0010644856,0.0006616678,0.10289684,0.42580834,0.017979877,0.0057143294,0.17628887],"study_design_scores_gemma":[0.00008074777,0.00025272588,0.21829322,0.00012915228,0.00075531355,0.0006693574,0.0005185734,0.64971155,0.06050579,0.05156322,0.017423982,0.000096370466],"about_ca_topic_score_codex":0.003979394,"about_ca_topic_score_gemma":0.0064855865,"teacher_disagreement_score":0.003979394,"about_ca_system_score_codex":0.00031017017,"about_ca_system_score_gemma":0.00041728432,"threshold_uncertainty_score":0.007912457},"labels":[],"label_agreement":null},{"id":"W4405855303","doi":"10.1186/s13040-024-00413-w","title":"Distinct network patterns emerge from Cartesian and XOR epistasis models: a comparative network science analysis","year":2024,"lang":"en","type":"article","venue":"BioData Mining","topic":"Bioinformatics and Genomic Networks","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"U.S. National Library of Medicine; National Institutes of Health; Alliance de recherche numérique du Canada","keywords":"Epistasis; Biology; Computational biology; Genetics; Gene","score_opus":0.03330849453372306,"score_gpt":0.27841169382494874,"score_spread":0.24510319929122568,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405855303","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.89748085,0.00035712818,0.09720111,0.00029563342,0.000016425214,0.00006853729,0.00084800867,0.00020372859,0.0035285463],"genre_scores_gemma":[0.9747277,0.00019171761,0.02369102,0.00003219237,0.000008253772,0.000076799646,0.00075490493,0.000020895219,0.0004965512],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9996604,0.00013960787,0.000013141306,0.00009568225,0.000052738356,0.000038493094],"domain_scores_gemma":[0.998621,0.00081443397,0.00024613965,0.000089733396,0.00014975156,0.00007891247],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009248908,0.0003773279,0.00032551156,0.0022823776,0.00042669466,0.00068217446,0.00037751286,0.0003480512,0.0021756897],"category_scores_gemma":[0.002863137,0.00015178374,0.00084241555,0.0009386529,0.00054605526,0.0008215741,0.0005246906,0.00041278338,0.00011368185],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016024275,0.00041432297,0.2258725,0.0011637037,0.002321658,0.0016427027,0.0022383344,0.38930252,0.10011742,0.15160167,0.0055945152,0.11812828],"study_design_scores_gemma":[0.000037531154,0.00015426106,0.07784018,0.000039808856,0.0002251105,0.00042768603,0.0005414815,0.8720633,0.003005221,0.043090705,0.002534257,0.000040469087],"about_ca_topic_score_codex":0.0031030288,"about_ca_topic_score_gemma":0.0036103723,"teacher_disagreement_score":0.0031030288,"about_ca_system_score_codex":0.0006771516,"about_ca_system_score_gemma":0.0004050562,"threshold_uncertainty_score":0.007278383},"labels":[],"label_agreement":null},{"id":"W4406160197","doi":"10.1186/s13040-024-00412-x","title":"The Venus score for the assessment of the quality and trustworthiness of biomedical datasets","year":2025,"lang":"en","type":"article","venue":"BioData Mining","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"HORIZON EUROPE European Research Council; Ministero dell'Università e della Ricerca; HORIZON EUROPE Framework Programme; Dipartimenti di Eccellenza; European Commission; Alexander von Humboldt-Stiftung","keywords":"Computer science; Quality (philosophy); Informatics; Usability; Data science; Data quality; Data mining; Artificial intelligence","score_opus":0.33500278629655134,"score_gpt":0.5494247029763654,"score_spread":0.21442191667981403,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406160197","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5358184,0.010626097,0.3203368,0.0041802027,0.0018464843,0.0038738702,0.054815438,0.0063841343,0.062118623],"genre_scores_gemma":[0.80751836,0.0011244464,0.15911233,0.00063445634,0.0002591086,0.0030549811,0.025090419,0.0009070197,0.0022988624],"study_design_codex":"observational","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9474834,0.019938018,0.009596187,0.004265033,0.017308393,0.0014090255],"domain_scores_gemma":[0.7859559,0.15146898,0.016693007,0.012226128,0.028531639,0.005124252],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.028480144,0.0012873128,0.0015990478,0.023185141,0.0020626443,0.004844851,0.0016444284,0.0020810626,0.0058767465],"category_scores_gemma":[0.21025762,0.00035333904,0.0022268794,0.016118914,0.0026802132,0.0051660887,0.005121651,0.0017871301,0.0018727232],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021947417,0.00038100246,0.4865975,0.0061901817,0.0028995632,0.0006445413,0.004277576,0.014191807,0.005135243,0.0374819,0.07453058,0.36547545],"study_design_scores_gemma":[0.0008650593,0.0027134418,0.4484774,0.0036355876,0.0018677048,0.0033939478,0.007798791,0.13523887,0.017766599,0.18429103,0.19227356,0.0016779667],"about_ca_topic_score_codex":0.0025044745,"about_ca_topic_score_gemma":0.0041978247,"teacher_disagreement_score":0.9715198,"about_ca_system_score_codex":0.0019173842,"about_ca_system_score_gemma":0.0025990014,"threshold_uncertainty_score":0.15061921},"labels":[],"label_agreement":null},{"id":"W4408930421","doi":"10.1186/s13040-025-00441-0","title":"Multivariate longitudinal clustering reveals neuropsychological factors as dementia predictors in an Alzheimer’s disease progression study","year":2025,"lang":"en","type":"article","venue":"BioData Mining","topic":"Dementia and Cognitive Impairment Research","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Multivariate statistics; Dementia; Neuropsychology; Multivariate analysis; Cluster analysis; Disease; Longitudinal study; Psychology; Alzheimer's disease; Longitudinal data; Medicine; Computer science; Cognition; Psychiatry; Artificial intelligence; Internal medicine; Data mining; Machine learning; Pathology","score_opus":0.10375330141926643,"score_gpt":0.43651103475654657,"score_spread":0.33275773333728015,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408930421","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9906781,0.00028878482,0.007431321,0.00020095876,0.000013444395,0.000046331992,0.0010621335,0.000052964795,0.00022586473],"genre_scores_gemma":[0.9912315,0.000102714665,0.006449226,0.000015894344,0.000009309887,0.000029842537,0.0020239342,0.000011187281,0.0001263266],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99913776,0.00043997873,0.000092824215,0.00015500054,0.00010664792,0.00006779128],"domain_scores_gemma":[0.99755305,0.001002427,0.0004474232,0.000388393,0.00038140424,0.00022740591],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041749226,0.00032407726,0.0004457108,0.0021633862,0.0004396564,0.0010209067,0.00047210456,0.00041312582,0.00063128857],"category_scores_gemma":[0.00792922,0.000128364,0.00086897536,0.0015868193,0.00019175629,0.00036249054,0.00078431575,0.00053630525,0.00019336984],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008227638,0.00019773391,0.949548,0.00006415777,0.00063933403,0.00023586814,0.00047407713,0.0106379315,0.0014574111,0.00046939435,0.001677534,0.033775806],"study_design_scores_gemma":[0.00005314937,0.00027473486,0.8235165,0.00006691693,0.00031401002,0.0003059174,0.0010316072,0.16894817,0.0010187963,0.0026978164,0.0017183148,0.00005418158],"about_ca_topic_score_codex":0.010699248,"about_ca_topic_score_gemma":0.013302905,"teacher_disagreement_score":0.010699248,"about_ca_system_score_codex":0.00043587066,"about_ca_system_score_gemma":0.0008883571,"threshold_uncertainty_score":0.022079349},"labels":[],"label_agreement":null},{"id":"W4410024327","doi":"10.1186/s13040-025-00444-x","title":"Learning the therapeutic targets of acute myeloid leukemia through multiscale human interactome network and community analysis","year":2025,"lang":"en","type":"article","venue":"BioData Mining","topic":"Computational Drug Discovery Methods","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Interactome; Myeloid leukemia; Computer science; Computational biology; Data science; Medicine; Biology; Cancer research; Gene; Genetics","score_opus":0.040764673387921384,"score_gpt":0.3493131891731776,"score_spread":0.3085485157852562,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410024327","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5324755,0.0030096683,0.4544201,0.0011201523,0.00005053467,0.00023735684,0.0041396995,0.0010145843,0.0035324674],"genre_scores_gemma":[0.91479224,0.0007782046,0.080781065,0.00010852106,0.000040326177,0.00014564232,0.0026462297,0.000031863536,0.0006758837],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999777,0.000069403126,0.000011525521,0.00007150488,0.00004190327,0.000028517967],"domain_scores_gemma":[0.999624,0.00020028667,0.00008044553,0.00002501952,0.000039633414,0.000030526546],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00039551695,0.0004323064,0.00047506252,0.0024570697,0.00034846435,0.00047305346,0.0003556337,0.0004304239,0.0008436393],"category_scores_gemma":[0.0015486851,0.00015990659,0.0007600026,0.0012327919,0.00023404698,0.0004865795,0.0006590908,0.00031977563,0.00011826536],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007801447,0.00058977073,0.088405006,0.00095047295,0.0008275843,0.00077959435,0.0003009202,0.5537912,0.033725478,0.017453391,0.010082311,0.29231405],"study_design_scores_gemma":[0.000017161668,0.0000485413,0.0074949656,0.000016158621,0.00008161849,0.000108779575,0.0000576169,0.97814894,0.0011966616,0.011286681,0.0015303155,0.000012493014],"about_ca_topic_score_codex":0.0054001193,"about_ca_topic_score_gemma":0.008050121,"teacher_disagreement_score":0.0054001193,"about_ca_system_score_codex":0.0005063793,"about_ca_system_score_gemma":0.0006849255,"threshold_uncertainty_score":0.0107373595},"labels":[],"label_agreement":null},{"id":"W4410588018","doi":"10.1186/s13040-025-00451-y","title":"Correction: Learning the therapeutic targets of acute myeloid leukemia through multiscale human interactome network and community analysis","year":2025,"lang":"en","type":"erratum","venue":"BioData Mining","topic":"Computational Drug Discovery Methods","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Interactome; Myeloid leukemia; Computer science; Computational biology; Artificial intelligence; Data science; Medicine; Biology; Cancer research; Gene","score_opus":0.03957594418586434,"score_gpt":0.33867440917041225,"score_spread":0.2990984649845479,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410588018","genre_codex":"editorial","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00016640553,0.00044217048,0.0011926506,0.056277953,0.93769854,0.000032034128,0.0021398063,0.0008165793,0.0012339473],"genre_scores_gemma":[0.030248506,0.005371784,0.016634762,0.2266866,0.4894493,0.00063804013,0.006076962,0.0046228664,0.22027116],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9933849,0.00097512134,0.0013440059,0.0010440603,0.0026071842,0.00064478046],"domain_scores_gemma":[0.93389374,0.018854028,0.002740388,0.0043858774,0.036932316,0.0031936537],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0055559124,0.003059515,0.0034030848,0.005065198,0.004893707,0.004438074,0.0051498665,0.011433324,0.09568404],"category_scores_gemma":[0.13085547,0.0019150404,0.002644032,0.0037768898,0.004257701,0.00290706,0.00344953,0.014041535,0.04872388],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005031353,0.0000076076326,0.00010605813,0.00015567502,0.000023573,0.00032592865,0.000034689416,0.000056095567,0.000056486486,0.0006165877,0.9949786,0.003588482],"study_design_scores_gemma":[0.00031674755,0.00005652468,0.0020271572,0.0007013693,0.00021297392,0.0021262788,0.00024766976,0.0015609831,0.0012282063,0.007569738,0.9838255,0.00012673522],"about_ca_topic_score_codex":0.026930362,"about_ca_topic_score_gemma":0.02517072,"teacher_disagreement_score":0.09568404,"about_ca_system_score_codex":0.0043573007,"about_ca_system_score_gemma":0.007060485,"threshold_uncertainty_score":0.320095},"labels":[],"label_agreement":null},{"id":"W4411237205","doi":"10.1186/s13040-025-00455-8","title":"DBSCAN and DBCV application to open medical records heterogeneous data for identifying clinically significant clusters of patients with neuroblastoma","year":2025,"lang":"en","type":"article","venue":"BioData Mining","topic":"Neuroblastoma Research and Treatments","field":"Medicine","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Data mining; Neuroblastoma; Medical record; DBSCAN; Data science; Information retrieval; Cluster analysis; Medicine; Internal medicine; Artificial intelligence; Fuzzy clustering","score_opus":0.07202205426547378,"score_gpt":0.3988111328466636,"score_spread":0.3267890785811898,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411237205","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.72025615,0.0025407725,0.16946718,0.0018470734,0.00042060163,0.0013104458,0.07551049,0.023228766,0.0054185134],"genre_scores_gemma":[0.64194477,0.0005970042,0.2625602,0.00025336648,0.00008254516,0.0008266542,0.09233971,0.00045403175,0.0009416436],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9957445,0.0011799394,0.0007573732,0.00088520773,0.001210504,0.00022244148],"domain_scores_gemma":[0.9924676,0.0037141263,0.00065871445,0.0010578921,0.0017685584,0.00033316447],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040159174,0.0009508963,0.0008787392,0.0073166564,0.00096550933,0.0021170487,0.0016243043,0.0009781744,0.0009563587],"category_scores_gemma":[0.020295328,0.00033819972,0.0011369056,0.006311735,0.000473189,0.0007578041,0.0018746341,0.0008984644,0.0004454311],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031117555,0.0010335657,0.19566879,0.0019339065,0.0014415119,0.001977312,0.0025104668,0.19293502,0.008032072,0.009757654,0.07291915,0.5086788],"study_design_scores_gemma":[0.0002557321,0.00020987856,0.054825243,0.00018402783,0.00023376898,0.00076408027,0.0021132454,0.8970829,0.010787505,0.009257781,0.024140006,0.00014576664],"about_ca_topic_score_codex":0.033274837,"about_ca_topic_score_gemma":0.028120298,"teacher_disagreement_score":0.033274837,"about_ca_system_score_codex":0.001145119,"about_ca_system_score_gemma":0.0027463732,"threshold_uncertainty_score":0.06616229},"labels":[],"label_agreement":null},{"id":"W4413362561","doi":"10.1186/s13040-025-00465-6","title":"A simple guide to the use of Student’s t-test, Mann-Whitney U test, Chi-squared test, and Kruskal-Wallis test in biostatistics","year":2025,"lang":"en","type":"article","venue":"BioData Mining","topic":"Genetics, Bioinformatics, and Biomedical Research","field":"Biochemistry, Genetics and Molecular Biology","cited_by":68,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Ministero dell'Università e della Ricerca; Dipartimenti di Eccellenza","keywords":"Test (biology); Biostatistics; Null hypothesis; Univariate; Kruskal–Wallis one-way analysis of variance; Mann–Whitney U test; Statistical hypothesis testing; Computer science; Chi-square test; Statistics; Simple (philosophy); Machine learning; Artificial intelligence; Data mining; Mathematics; Multivariate statistics; Medicine","score_opus":0.04371332968964679,"score_gpt":0.34307389137528277,"score_spread":0.29936056168563596,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413362561","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0022455568,0.024614943,0.7550923,0.046200626,0.016263204,0.0049807415,0.025603138,0.023069765,0.10192976],"genre_scores_gemma":[0.00891192,0.017465767,0.887909,0.01940783,0.0045575704,0.007259248,0.010207097,0.0072864643,0.036995057],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9735796,0.013683006,0.0034571586,0.0016913,0.0070934393,0.0004954746],"domain_scores_gemma":[0.89549845,0.06984838,0.0048049325,0.006376815,0.021514291,0.00195706],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018384218,0.00255582,0.002002,0.009598304,0.0018151229,0.0044577573,0.004592329,0.005096974,0.08368386],"category_scores_gemma":[0.13586693,0.0017611461,0.0017375058,0.01084566,0.004148006,0.0058513973,0.003290198,0.009000804,0.09816351],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000122049576,0.00018874658,0.00088638766,0.0015523406,0.000045526285,0.00039233392,0.0006732313,0.0009575,0.0016128023,0.028731802,0.81350183,0.15133548],"study_design_scores_gemma":[0.00005276039,0.00014266706,0.00093921437,0.0013155936,0.000018044053,0.0006402922,0.00037494246,0.0012509662,0.0008696135,0.039582152,0.95470446,0.00010929445],"about_ca_topic_score_codex":0.0039617056,"about_ca_topic_score_gemma":0.005516058,"teacher_disagreement_score":0.08368386,"about_ca_system_score_codex":0.0021365576,"about_ca_system_score_gemma":0.007232973,"threshold_uncertainty_score":0.27995044},"labels":[],"label_agreement":null},{"id":"W4414531239","doi":"10.1186/s13040-025-00480-7","title":"Temporal phenotyping and prognostic stratification of patients with sepsis through longitudinal clustering","year":2025,"lang":"en","type":"article","venue":"BioData Mining","topic":"Time Series Analysis and Forecasting","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Cluster analysis; Multivariate statistics; Sepsis; Covariate; Silhouette; Disease; Biobank; Multivariate analysis; Metric (unit)","score_opus":0.0254430725212012,"score_gpt":0.24562285390312613,"score_spread":0.22017978138192493,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414531239","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.81801283,0.0015371455,0.17615928,0.00089307176,0.00010178739,0.00013709617,0.0016127607,0.0005501703,0.0009957852],"genre_scores_gemma":[0.96808004,0.00032974358,0.029165959,0.00006747918,0.000055049903,0.0000694931,0.0017368556,0.00003443933,0.0004609162],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992473,0.00032977783,0.000059046957,0.00018679061,0.00009123098,0.000085845415],"domain_scores_gemma":[0.997811,0.0007549901,0.00046542895,0.00032206808,0.00045941424,0.00018714843],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025613203,0.0005546821,0.00062009954,0.0016329079,0.00046197185,0.0007556882,0.0006678177,0.00054022577,0.0006441487],"category_scores_gemma":[0.0061383974,0.00022247554,0.0008220017,0.0011223551,0.00022828238,0.00049830787,0.0008932034,0.0007391527,0.00032104942],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0019564857,0.00043477627,0.6724594,0.00015921688,0.0006392748,0.0004512558,0.00075946783,0.15922925,0.007390224,0.0024627033,0.0054727416,0.14858514],"study_design_scores_gemma":[0.000037919403,0.00021053774,0.083908,0.00004953392,0.00016446857,0.00026632793,0.00030710542,0.907019,0.0021240665,0.0041219927,0.0017202921,0.00007073064],"about_ca_topic_score_codex":0.007898129,"about_ca_topic_score_gemma":0.0062728617,"teacher_disagreement_score":0.007898129,"about_ca_system_score_codex":0.0005530801,"about_ca_system_score_gemma":0.0009460476,"threshold_uncertainty_score":0.015704274},"labels":[],"label_agreement":null},{"id":"W4414606898","doi":"10.1186/s13040-025-00482-5","title":"Proteome mining of Yersinia Enterocolitica for drug targets and computational inhibitor identification with ADMET, anti-inflammation potential and formulation characteristics","year":2025,"lang":"en","type":"article","venue":"BioData Mining","topic":"Pharmacological Effects of Natural Compounds","field":"Pharmacology, Toxicology and Pharmaceutics","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Alpha Technologies (Canada)","funders":"King Khalid University; National Research Foundation of Korea","keywords":"Yersinia enterocolitica; Proteome; Drug; Yersinia; Proteomics; Drug repositioning; Identification (biology); Drug discovery","score_opus":0.04233579431491202,"score_gpt":0.38111179151165103,"score_spread":0.338775997196739,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414606898","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.85045797,0.012444383,0.045627747,0.001688677,0.00014064721,0.00041154408,0.072548114,0.00787473,0.008806202],"genre_scores_gemma":[0.80358595,0.00698246,0.08076717,0.00046101844,0.000052225034,0.00048149348,0.103958264,0.00037964154,0.003331787],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9998153,0.000028030065,0.000016839569,0.00006128697,0.00004889691,0.00002971424],"domain_scores_gemma":[0.9998418,0.00005149517,0.000035644094,0.00001221615,0.00004398902,0.000014825872],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004617121,0.0011831034,0.001233775,0.0011758092,0.0004677394,0.0009491267,0.0006324538,0.00066094904,0.0019603278],"category_scores_gemma":[0.00073494954,0.00031652985,0.0017722583,0.0014820196,0.00014483766,0.000458112,0.00033058232,0.00047211713,0.00071848114],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0038264664,0.0016388365,0.07522388,0.010800095,0.0018741713,0.003839357,0.0004194461,0.33240563,0.32367113,0.0055000084,0.043352183,0.19744882],"study_design_scores_gemma":[0.00031668064,0.0013476303,0.026553882,0.00020781896,0.0010280864,0.0013905196,0.0003729727,0.8402478,0.078319155,0.003538462,0.04657304,0.0001039012],"about_ca_topic_score_codex":0.0038879947,"about_ca_topic_score_gemma":0.0042632986,"teacher_disagreement_score":0.0038879947,"about_ca_system_score_codex":0.0006286438,"about_ca_system_score_gemma":0.0014095844,"threshold_uncertainty_score":0.0077307224},"labels":[],"label_agreement":null},{"id":"W4416785648","doi":"10.1186/s13040-025-00499-w","title":"Bioinformatics analysis of Rickettsia typhi autoimmune associations and screening of Streptomyces-derived inhibitors","year":2025,"lang":"en","type":"article","venue":"BioData Mining","topic":"vaccines and immunoinformatics approaches","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Alpha Technologies (Canada)","funders":"National Research Foundation of Korea; King Khalid University; National Research Foundation","keywords":"Rickettsia typhi; Salmonella typhi; Rickettsia","score_opus":0.018265540243988583,"score_gpt":0.2611037845670206,"score_spread":0.24283824432303203,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416785648","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8920094,0.015087206,0.021570912,0.0009935553,0.00019472497,0.0004610868,0.056005694,0.0054219,0.008255474],"genre_scores_gemma":[0.80089927,0.008152815,0.069915265,0.00050249766,0.000066183886,0.00051164726,0.116882615,0.00034178817,0.002727955],"study_design_codex":"bench_or_experimental","study_design_gemma":"not_applicable","domain_scores_codex":[0.99963236,0.000058087742,0.00004699797,0.0001057031,0.00010017394,0.00005671828],"domain_scores_gemma":[0.9997192,0.00007691282,0.000073810894,0.000014714613,0.000075047355,0.000040328785],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0005269859,0.00093529304,0.0014504092,0.0020261267,0.00046157575,0.0010485128,0.00045899188,0.00040891947,0.0018627555],"category_scores_gemma":[0.00077525136,0.00018884066,0.001593284,0.002042684,0.00014065544,0.00035104767,0.00036241504,0.00043862054,0.0009760379],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0068384744,0.0019998334,0.17517538,0.012585054,0.0023523786,0.0056103347,0.00052962045,0.05072153,0.43859875,0.0027479804,0.051248513,0.2515922],"study_design_scores_gemma":[0.000633825,0.0039979876,0.25077313,0.0007230168,0.003098752,0.004989701,0.00115025,0.43167198,0.17972358,0.0040349932,0.11889474,0.00030801323],"about_ca_topic_score_codex":0.0013549693,"about_ca_topic_score_gemma":0.002012404,"teacher_disagreement_score":0.0020261267,"about_ca_system_score_codex":0.00048955745,"about_ca_system_score_gemma":0.00081680575,"threshold_uncertainty_score":0.0062315464},"labels":[],"label_agreement":null}]}