{"id":"W6977553122","doi":"10.6084/m9.figshare.c.7248780","title":"Testing the generalizability and effectiveness of deep learning models among clinics: sperm detection as a pilot study","year":2024,"lang":"en","type":"other","venue":"Figshare","topic":"DNA and Biological Computing","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"CReATe Fertility Centre; University of Toronto","funders":"","keywords":"Deep learning; Generalizability theory; Preprocessor; Sample (material); Intraclass correlation; Object detection; Pattern recognition (psychology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002977324,0.0001956805,0.0002181614,0.00002954509,0.00005437573,0.00003393507,0.0001535291,0.0001783856,0.0008717308],"category_scores_gemma":[0.001037746,0.0001280514,0.00006054198,0.00009324412,0.00002666189,0.000001505653,0.0003540155,0.0002414195,0.00002988033],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000007682252,"about_ca_system_score_gemma":0.00001686692,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007429255,"about_ca_topic_score_gemma":0.00005195128,"domain_scores_codex":[0.9987482,0.000374138,0.0001745303,0.0004743719,0.00008656309,0.0001422315],"domain_scores_gemma":[0.9994085,0.0001310452,0.0001499853,0.0002174137,0.00005977107,0.00003330176],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"observational","study_design_scores_codex":[0.002241475,0.003532295,0.09310941,0.02662977,0.00467091,0.0002092609,0.0009779881,0.00957851,0.4307368,0.0000343221,0.04422766,0.3840516],"study_design_scores_gemma":[0.009259302,0.1339442,0.4479777,0.04152507,0.001961917,0.0002690827,0.003028239,0.07780217,0.05364637,0.005081096,0.2155839,0.00992101],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8905047,0.0104787,0.000117741,0.000002976429,0.0002737797,0.002797182,0.007309429,0.0002407463,0.08827473],"genre_scores_gemma":[0.9954035,0.000006982412,0.00002542449,0.00001115371,0.0002640515,0.00008171993,0.002154131,0.0001023895,0.001950647],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3770905,"threshold_uncertainty_score":0.954484,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06236278408729332,"score_gpt":0.2965040417806014,"score_spread":0.2341412576933081,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}