{"id":"W4297996437","doi":"10.3389/fonc.2022.979336","title":"Functional and embedding feature analysis for pan-cancer classification","year":2022,"lang":"en","type":"article","venue":"Frontiers in Oncology","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Chinese Academy of Sciences","keywords":"Feature selection; Computer science; Feature (linguistics); Artificial intelligence; Machine learning; KEGG; Word2vec; Pattern recognition (psychology); Computational biology; Embedding; Biology; Genetics; Gene ontology; Gene","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001022968,0.0008207572,0.0008207314,0.003408035,0.0003565071,0.0006639441,0.0006255851,0.0005561702,0.002634807],"category_scores_gemma":[0.002101378,0.0001326604,0.0009503446,0.002326572,0.0002449791,0.0005545798,0.000617763,0.0006216573,0.0009754489],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004077814,"about_ca_system_score_gemma":0.000554356,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002514119,"about_ca_topic_score_gemma":0.0019812,"domain_scores_codex":[0.999458,0.0001249273,0.00005230772,0.0001572527,0.0001242698,0.00008319352],"domain_scores_gemma":[0.9993124,0.0002825392,0.0000825063,0.000104619,0.000180731,0.00003716904],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006384174,0.0004308915,0.02717841,0.0004921991,0.0003183023,0.0004215341,0.0001263268,0.05356701,0.04326833,0.005480705,0.0146094,0.8534684],"study_design_scores_gemma":[0.00004412834,0.0003264314,0.02470026,0.00005969356,0.0001706337,0.0004164964,0.0001308451,0.9326002,0.01377664,0.01663922,0.01106934,0.00006620437],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1867081,0.003617043,0.7918901,0.0005466448,0.0001674354,0.0002319333,0.008096728,0.005956421,0.002785557],"genre_scores_gemma":[0.7912942,0.0006702106,0.1908397,0.0001162943,0.0001170453,0.0003773192,0.01411465,0.000187648,0.002282953],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.003408035,"threshold_uncertainty_score":0.008814275,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01525595278101351,"score_gpt":0.3094189619132441,"score_spread":0.2941630091322306,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}