{"id":"W4413312092","doi":"10.1073/pnas.2506316122","title":"Sparse autoencoders uncover biologically interpretable features in protein language model representations","year":2025,"lang":"en","type":"article","venue":"Proceedings of the National Academy of Sciences","topic":"Machine Learning in Bioinformatics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":25,"is_retracted":false,"has_abstract":true,"ca_institutions":"Toronto Metropolitan University","funders":"National Institute of General Medical Sciences; National Cancer Institute; Computer Science and Artificial Intelligence Laboratory, Massachusetts Institute of Technology; National Institutes of Health; Massachusetts Institute of Technology","keywords":"Computer science; Artificial intelligence; Natural language processing; Pattern recognition (psychology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007690613,0.0005926439,0.0003506847,0.0005188299,0.0001792521,0.0006426068,0.0004573907,0.0005495143,0.0005626284],"category_scores_gemma":[0.004357098,0.0003619793,0.000672807,0.0003258316,0.000730129,0.001131606,0.0009724429,0.001705176,0.0002908181],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003953394,"about_ca_system_score_gemma":0.0004492306,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002327676,"about_ca_topic_score_gemma":0.003482941,"domain_scores_codex":[0.9996631,0.000130455,0.00001656551,0.0000803785,0.0000704629,0.00003910445],"domain_scores_gemma":[0.9981792,0.001295923,0.0001496816,0.000199125,0.0001374302,0.00003866971],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001861859,0.0001269076,0.00699902,0.0001646206,0.0001649383,0.0002737884,0.0005458971,0.6945035,0.06231336,0.03327709,0.00335226,0.1980924],"study_design_scores_gemma":[0.000004066178,0.00002138484,0.0008577088,0.000006581749,0.000007314914,0.00002039471,0.00002452187,0.9810535,0.002790069,0.01482382,0.0003840151,0.000006594975],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1505904,0.0002484413,0.8462792,0.0004824894,0.00003014021,0.0000249273,0.0003110145,0.0008097156,0.001223558],"genre_scores_gemma":[0.8660067,0.000319091,0.1297925,0.0002772458,0.00005282706,0.00008425317,0.001164804,0.0001553733,0.002147159],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.002327676,"threshold_uncertainty_score":0.004628241,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01540583249850305,"score_gpt":0.3230372084270323,"score_spread":0.3076313759285293,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}