{"id":"W4392503529","doi":"10.2214/ajr.24.31060","title":"Reply to “Zero-, Single-, and Few-Shot Learning in Large Language Models to Identify Incidental Findings From Radiology Reports”","year":2024,"lang":"en","type":"letter","venue":"American Journal of Roentgenology","topic":"Radiology practices and education","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Princess Margaret Cancer Centre; University of Toronto; University Health Network","funders":"","keywords":"Medicine; Zero (linguistics); Radiology; Medical physics; Shot (pellet); Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.0008528547,0.0003201337,0.001319385,0.001233441,0.00005635408,0.00004594566,0.0002038895,0.0003667076,0.0001155084],"category_scores_gemma":[0.0006803226,0.0002942065,0.0001658075,0.0004128363,0.0001816632,0.0001604344,0.0001626046,0.003125125,0.00004629038],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004441812,"about_ca_system_score_gemma":0.0002462871,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001324204,"about_ca_topic_score_gemma":0.00005573824,"domain_scores_codex":[0.9969656,0.000403225,0.001109625,0.0006351888,0.0002701431,0.0006162486],"domain_scores_gemma":[0.9979722,0.0003466121,0.0009555619,0.0003606618,0.0001273596,0.0002375611],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000836302,0.0001486888,0.05595033,0.00008295328,0.0009503561,0.04476072,0.02546937,0.0002611927,0.01388736,0.000004958985,0.8514016,0.006246174],"study_design_scores_gemma":[0.001653275,0.01163562,0.05206772,0.001080778,0.001502222,0.111557,0.02172637,0.0001015661,0.000301056,0.0007234581,0.7967676,0.0008833461],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6941055,0.001596405,0.0002745295,0.3025266,0.001159739,0.0001887431,0.000009210226,0.00001913045,0.0001201357],"genre_scores_gemma":[0.6714709,0.000232997,0.001471575,0.3243351,0.001590311,0.00001119229,0.0001027697,0.00006569696,0.0007194715],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.06679627,"threshold_uncertainty_score":0.999951,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02833361801041726,"score_gpt":0.3311129402821396,"score_spread":0.3027793222717223,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}