{"id":"W4416437085","doi":"10.48550/arxiv.2511.02824","title":"Kosmos: An AI Scientist for Autonomous Discovery","year":2025,"lang":"","type":"preprint","venue":"ArXiv.org","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institutes of Health; Engineering and Physical Sciences Research Council; UK Dementia Research Institute; Medical Research Council; Cure Alzheimer's Fund; Canadian Institute for Advanced Research; Toyota Research Institute","keywords":"Limiting; Code (set theory); Reading (process); Scientific discovery; Source code","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007037536,0.001136109,0.0008178171,0.002065377,0.001216642,0.004322555,0.003692292,0.001823326,0.01276967],"category_scores_gemma":[0.02803049,0.001195533,0.001519148,0.001443997,0.002453166,0.007248043,0.006850749,0.003494213,0.008162281],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001097783,"about_ca_system_score_gemma":0.003920333,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0013254,"about_ca_topic_score_gemma":0.002266352,"domain_scores_codex":[0.9960915,0.001169948,0.0003140655,0.0008092302,0.001371452,0.0002437436],"domain_scores_gemma":[0.9855843,0.00619288,0.0007503111,0.003922323,0.002248126,0.001302196],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001717796,0.0005702223,0.009871294,0.001828633,0.0006446128,0.001300439,0.003973131,0.04343051,0.03629002,0.215499,0.247892,0.4369823],"study_design_scores_gemma":[0.0004952537,0.000320439,0.001306833,0.0002915411,0.0001789784,0.0006874757,0.0005685031,0.2735191,0.02327898,0.1976646,0.5015136,0.0001747983],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"software","genre_scores_codex":[0.01694567,0.00108613,0.8563376,0.005718736,0.0009936459,0.000706844,0.002352605,0.08105561,0.0348031],"genre_scores_gemma":[0.1102207,0.001235639,0.8568304,0.001998524,0.0003067839,0.001072606,0.004933935,0.006785587,0.01661581],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.01276967,"threshold_uncertainty_score":0.04271883,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03738130673923349,"score_gpt":0.333183693005038,"score_spread":0.2958023862658045,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}