{"id":"W4408068147","doi":"10.1007/s00467-025-06729-x","title":"Evaluating large language models in pediatric nephrology","year":2025,"lang":"en","type":"editorial","venue":"Pediatric Nephrology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"Children's Hospital of Western Ontario; Western University","funders":"","keywords":"Nephrology; Medicine; Internal medicine; Intensive care medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","research_integrity"],"consensus_categories":["research_integrity"],"category_scores_codex":[0.002240394,0.0005672261,0.001315993,0.001909806,0.0001658007,0.00002830612,0.0004316934,0.002991973,0.0004079186],"category_scores_gemma":[0.003917536,0.0005883446,0.0002505655,0.001553073,0.00006607643,0.0001402425,0.0002341732,0.002927606,0.0002760704],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004662811,"about_ca_system_score_gemma":0.004869196,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001874371,"about_ca_topic_score_gemma":0.001152674,"domain_scores_codex":[0.9944712,0.0005827384,0.001598605,0.001191084,0.0007888171,0.001367567],"domain_scores_gemma":[0.9939817,0.00371532,0.0005559012,0.0008484972,0.00064149,0.0002570897],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0005208898,0.0003969804,0.007856803,0.001102722,0.0000409017,0.0001366078,0.001680468,0.0001737213,0.00001263796,0.0001170254,0.9821302,0.005831038],"study_design_scores_gemma":[0.003256759,0.004941721,0.005164384,0.00016045,0.00621443,0.00007762718,0.001738716,0.0198526,0.00004458232,0.01658569,0.939368,0.002595057],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"editorial","genre_gemma":"editorial","genre_scores_codex":[0.243718,0.0157303,0.0004345764,0.002428793,0.7339638,0.001800848,0.0001525052,0.000236476,0.001534797],"genre_scores_gemma":[0.03110118,0.01138513,0.000825491,0.002940717,0.9498153,0.0004188664,0.001266494,0.00009418868,0.002152625],"genre_candidate":"editorial","genre_consensus":"editorial","teacher_disagreement_score":0.2158515,"threshold_uncertainty_score":0.9996568,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07570928681527708,"score_gpt":0.440631592931464,"score_spread":0.3649223061161869,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}