{"id":"W7117851814","doi":"","title":"Galaxy Zoo Evo: 1 million human-annotated images of galaxies","year":2025,"lang":"","type":"article","venue":"ArXiv.org","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; University of Toronto; Alfred P. Sloan Foundation; National Science Foundation","keywords":"Galaxy; Space (punctuation); Benchmark (surveying); Image (mathematics); Domain (mathematical analysis); Spiral galaxy","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001220018,0.002907885,0.001137031,0.003451633,0.001436208,0.002144764,0.003768561,0.003956403,0.01366025],"category_scores_gemma":[0.005731757,0.0009037761,0.001844224,0.003677871,0.001304924,0.001982682,0.003143864,0.002149135,0.01604727],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002287326,"about_ca_system_score_gemma":0.001796307,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.05138827,"about_ca_topic_score_gemma":0.1182628,"domain_scores_codex":[0.9983083,0.0002646309,0.00009476006,0.0006851489,0.0004352437,0.0002119176],"domain_scores_gemma":[0.9979261,0.000434704,0.0001790204,0.0008045956,0.0004256104,0.0002299185],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0007065277,0.0003652766,0.009793621,0.002389684,0.0003312256,0.0004508926,0.0003307436,0.007521422,0.003417508,0.001999992,0.9157099,0.05698323],"study_design_scores_gemma":[0.0007289426,0.0004219527,0.04638048,0.001239044,0.0002487148,0.002029007,0.0009922893,0.06773602,0.01058569,0.01093763,0.8584336,0.0002665227],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.03544241,0.004801462,0.01092463,0.001360153,0.0006592381,0.0006555812,0.9084631,0.02348422,0.01420921],"genre_scores_gemma":[0.02932266,0.0004026229,0.01724018,0.0004393625,0.00006802264,0.0003001091,0.9486219,0.0008426453,0.002762359],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.05138827,"threshold_uncertainty_score":0.1021783,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03054621689064294,"score_gpt":0.3079074448161753,"score_spread":0.2773612279255323,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}