{"id":"W6958498781","doi":"10.6084/m9.figshare.25884981","title":"Additional file 1 of Testing the generalizability and effectiveness of deep learning models among clinics: sperm detection as a pilot study","year":2024,"lang":"en","type":"article","venue":"Figshare","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"CReATe Fertility Centre; University of Toronto","funders":"","keywords":"Generalizability theory; Deep learning; Calibration; Artificial neural network; Reliability (semiconductor)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.003131336,0.000971447,0.000906675,0.001356372,0.001146515,0.00138455,0.001877215,0.00147282,0.8678674],"category_scores_gemma":[0.1019976,0.0005183442,0.001026283,0.001591665,0.0003422075,0.001500563,0.0007860186,0.001203653,0.1063136],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009120366,"about_ca_system_score_gemma":0.002057172,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00666339,"about_ca_topic_score_gemma":0.009964971,"domain_scores_codex":[0.9988981,0.0004392289,0.000171807,0.0002170131,0.000191162,0.00008258993],"domain_scores_gemma":[0.8808365,0.1086417,0.001947862,0.002994367,0.004731822,0.0008477584],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0016077,0.0006556068,0.003982116,0.004045634,0.0001392459,0.00008126473,0.0001580939,0.0008602282,0.00009594384,0.001233094,0.9610237,0.02611744],"study_design_scores_gemma":[0.06130091,0.005597271,0.09346677,0.01076186,0.001558137,0.001738695,0.002593025,0.01417738,0.002889817,0.04195087,0.7633947,0.000570551],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"other","genre_scores_codex":[0.002545048,0.00008599352,0.001993156,0.0005570524,0.0001270125,0.001297323,0.9896691,0.0006549455,0.003070397],"genre_scores_gemma":[0.09822825,0.0005381987,0.03267151,0.003944138,0.0004823738,0.06338941,0.7386904,0.002943967,0.05911183],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.8678674,"threshold_uncertainty_score":0.1884711,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2761740557161022,"score_gpt":0.4078399816966094,"score_spread":0.1316659259805072,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}