{"id":"W4416660071","doi":"10.2196/76896","title":"Evaluating Locally Run Large Language Models (Gemma 2, Mistral Nemo, and Llama 3) for Outpatient Otorhinolaryngology Care: Retrospective Study","year":2025,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Documentation; Retrospective cohort study; Otorhinolaryngology; MEDLINE; Outpatient clinic","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01371681,0.0007032989,0.0009882423,0.00118309,0.0004758142,0.001263594,0.0009760216,0.0007475843,0.001497122],"category_scores_gemma":[0.05871582,0.0004501862,0.001894962,0.0008172713,0.0009133984,0.001572684,0.001522917,0.001065255,0.0005123656],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001468423,"about_ca_system_score_gemma":0.001978025,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001900759,"about_ca_topic_score_gemma":0.002581121,"domain_scores_codex":[0.9893392,0.007041797,0.001265453,0.000850837,0.001182103,0.0003205397],"domain_scores_gemma":[0.9484866,0.03260154,0.008086478,0.003593178,0.004499923,0.00273242],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00954124,0.004859294,0.90548,0.00131871,0.0007275529,0.001141463,0.006956012,0.004407396,0.001274116,0.0002378438,0.001383257,0.06267312],"study_design_scores_gemma":[0.002016741,0.08264086,0.7991266,0.001158401,0.002741028,0.004708742,0.0238355,0.06923043,0.005662047,0.001076361,0.007323188,0.0004802195],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9979918,0.0002151718,0.0009274362,0.00005247268,0.00001286294,0.0003396258,0.0002195962,0.00002856448,0.0002124753],"genre_scores_gemma":[0.9960222,0.0001362208,0.002478249,0.00008631837,0.00001526331,0.0005897699,0.0005534086,0.00002184832,0.00009677687],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01371681,"threshold_uncertainty_score":0.07254225,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2242344059010077,"score_gpt":0.5599791578838738,"score_spread":0.335744751982866,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}