{"id":"W2953087206","doi":"10.48550/arxiv.1906.07282","title":"The Cells Out of Sample (COOS) dataset and benchmarks for measuring out-of-sample generalization of image classifiers","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Cell Image Analysis Techniques","field":"Biochemistry, Genetics and Molecular Biology","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Sample (material); Generalization; Artificial intelligence; Pattern recognition (psychology); Computer science; Image (mathematics); Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003432069,0.0001890563,0.0003232681,0.0000912261,0.00005702407,0.00001681952,0.0004751164,0.0002371792,0.00001063788],"category_scores_gemma":[0.0001826739,0.0001853604,0.0001986903,0.00007512498,0.0002359536,0.000009942086,0.0006809192,0.000107997,3.077453e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002326819,"about_ca_system_score_gemma":0.00008958838,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003426375,"about_ca_topic_score_gemma":0.0003438036,"domain_scores_codex":[0.9988865,0.00008739422,0.0002868618,0.0005090443,0.00006541809,0.0001647611],"domain_scores_gemma":[0.9980783,0.000132395,0.0005223169,0.000940871,0.0002813537,0.0000447752],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0003048658,0.0001229221,0.008919633,0.0007183766,0.0007308926,0.000002008813,0.0001239125,0.01076157,0.9642459,0.0008078211,0.0128602,0.0004019086],"study_design_scores_gemma":[0.0004804685,0.0001359752,0.0001522858,0.00005627427,0.0004315102,1.421739e-7,0.0001576015,0.01291128,0.9730552,0.001751929,0.01056679,0.0003005641],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4337513,0.0002371321,0.5616218,0.00001173786,0.0001343833,0.000630527,0.003263384,0.000009504918,0.0003401601],"genre_scores_gemma":[0.9890533,0.001649778,0.003359264,0.00001147604,0.00003561842,0.000001828682,0.005715024,0.00002003719,0.0001536661],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5582626,"threshold_uncertainty_score":0.7558777,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05735388509598731,"score_gpt":0.2279074514219108,"score_spread":0.1705535663259234,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}