{"id":"W4415320742","doi":"10.1016/j.jbi.2025.104932","title":"Towards a Biological Evaluation Framework for Oversampling (BEFO) gene expression data","year":2025,"lang":"en","type":"article","venue":"Journal of Biomedical Informatics","topic":"Gene expression and cancer classification","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Queen's University; Queen's University Belfast","keywords":"Oversampling; Ranking (information retrieval); Random forest; Biological data; Relevance (law); Synthetic data; Population; Class (philosophy)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0306565,0.001082678,0.001291704,0.002513832,0.0007195189,0.002778864,0.001737219,0.00137,0.0006073357],"category_scores_gemma":[0.06470821,0.000370139,0.001131267,0.001320727,0.002057203,0.002291595,0.002444214,0.001903216,0.0001963318],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001631032,"about_ca_system_score_gemma":0.001745379,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002756745,"about_ca_topic_score_gemma":0.001941924,"domain_scores_codex":[0.9809018,0.01142767,0.001296628,0.001502768,0.004396069,0.0004749944],"domain_scores_gemma":[0.966715,0.0197089,0.003016086,0.0031382,0.006791298,0.0006304884],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008580618,0.0003291661,0.02406113,0.0006102079,0.0003390066,0.0003357217,0.0006846039,0.4841968,0.03878579,0.07963922,0.004657267,0.3655031],"study_design_scores_gemma":[0.00004565142,0.0003970152,0.004015835,0.00008440956,0.00005261762,0.0001895562,0.0001399608,0.9370165,0.01180362,0.04184351,0.004354965,0.00005643744],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02173444,0.0006556104,0.9759718,0.0004017263,0.00005781028,0.0001326763,0.0001569541,0.0003480803,0.0005410126],"genre_scores_gemma":[0.3494684,0.0004691273,0.6471686,0.000515999,0.0001970045,0.0005054676,0.001048406,0.0001509987,0.0004759423],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0306565,"threshold_uncertainty_score":0.162129,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09341193759917463,"score_gpt":0.4020765994663315,"score_spread":0.3086646618671569,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}