{"id":"W4398216167","doi":"10.1186/s12958-024-01232-8","title":"Testing the generalizability and effectiveness of deep learning models among clinics: sperm detection as a pilot study","year":2024,"lang":"en","type":"article","venue":"Reproductive Biology and Endocrinology","topic":"Reproductive Biology and Fertility","field":"Medicine","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"CReATe Fertility Centre; University of Toronto","funders":"National Key Research and Development Program of China; Shenzhen Science and Technology Innovation Program; Basic and Applied Basic Research Foundation of Guangdong Province; Chinese University of Hong Kong; National Natural Science Foundation of China; Chinese University of Hong Kong, Shenzhen","keywords":"Generalizability theory; Artificial intelligence; Computer science; Deep learning; Preprocessor; Sample (material); Machine learning; Intraclass correlation; Object detection; Pattern recognition (psychology); Statistics; Reproducibility; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01544657,0.001252113,0.0009058555,0.0006748703,0.0005611861,0.000982583,0.001420359,0.001814981,0.001284605],"category_scores_gemma":[0.05597881,0.0006195912,0.001752312,0.0004103538,0.001536323,0.001130769,0.001956647,0.002208235,0.000721459],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007680269,"about_ca_system_score_gemma":0.0009879101,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004081255,"about_ca_topic_score_gemma":0.002531321,"domain_scores_codex":[0.9937307,0.002500982,0.0004866355,0.002270001,0.0007424029,0.0002692829],"domain_scores_gemma":[0.9560955,0.02833326,0.002199067,0.008859842,0.003720168,0.0007921457],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.01683031,0.006490328,0.3185549,0.001594809,0.004064576,0.0009269654,0.003036394,0.2401874,0.1087131,0.002068141,0.008593142,0.2889399],"study_design_scores_gemma":[0.001518796,0.02737359,0.1938896,0.0002158969,0.002212098,0.001366064,0.001078346,0.6631104,0.09723558,0.006051281,0.005612768,0.0003356029],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9517156,0.000407675,0.04475823,0.0003139722,0.0001306194,0.0006102034,0.0006162904,0.0005837288,0.000863737],"genre_scores_gemma":[0.970441,0.0001164485,0.02671683,0.0002941285,0.00005036381,0.0006047925,0.001090209,0.0001077042,0.0005785717],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01544657,"threshold_uncertainty_score":0.08169025,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06820193175808054,"score_gpt":0.3389963059083316,"score_spread":0.270794374150251,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}