{"id":"W4410729271","doi":"10.21203/rs.3.rs-6667194/v1","title":"Cell Behavior Video Classification Challenge, a benchmark for computer vision methods in live-cell imaging","year":2025,"lang":"en","type":"preprint","venue":"Research Square","topic":"Cell Image Analysis Techniques","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"Canadian Institutes of Health Research; Natural Sciences and Engineering Research Council of Canada; Nederlandse Organisatie voor Wetenschappelijk Onderzoek; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung; National Science Foundation","keywords":"Computer science; Benchmark (surveying); Artificial intelligence; Benchmarking; Tracking (education); Computer vision; Field (mathematics); Video tracking; Beach morphodynamics; Pattern recognition (psychology); Video processing","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002854009,0.002820011,0.00174933,0.002604881,0.0015074,0.002188517,0.003460479,0.003627087,0.004540403],"category_scores_gemma":[0.007834949,0.0004701418,0.001394874,0.002518041,0.0006641134,0.001524612,0.002148175,0.002506853,0.005192143],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001468618,"about_ca_system_score_gemma":0.002197346,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02177579,"about_ca_topic_score_gemma":0.02210405,"domain_scores_codex":[0.9977508,0.0004008388,0.0001243968,0.0006225667,0.000837157,0.0002642673],"domain_scores_gemma":[0.9959877,0.00134794,0.0001730768,0.0008242216,0.001197679,0.0004693117],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001220521,0.001391005,0.003783186,0.001706418,0.0003788186,0.0004458805,0.0001601498,0.02641862,0.02835136,0.004061455,0.4701431,0.4619394],"study_design_scores_gemma":[0.0005962301,0.001028955,0.02306007,0.0002516927,0.0002303595,0.001770155,0.0005118676,0.7417573,0.06329808,0.01343304,0.1539408,0.0001213446],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.36624,0.0274225,0.2827064,0.01651235,0.006194549,0.002934701,0.1822608,0.0580889,0.0576397],"genre_scores_gemma":[0.2392929,0.004469261,0.3218229,0.001688821,0.0009137525,0.001509674,0.3852311,0.003006412,0.04206507],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02177579,"threshold_uncertainty_score":0.04329807,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05001304550047966,"score_gpt":0.4729344888677439,"score_spread":0.4229214433672642,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}