{"id":"W4391983664","doi":"10.5114/pja.2024.135380","title":"Evaluating ChatGPT-3.5 in allergology: performance in the Polish Specialist Examination","year":2024,"lang":"en","type":"article","venue":"Alergologia Polska - Polish Journal of Allergology","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":10,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Library science; Computer science","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004531548,0.0002843078,0.0006332173,0.001160865,0.0001125328,0.00007611321,0.0005073003,0.0004439919,0.000308911],"category_scores_gemma":[0.001699441,0.0002111588,0.000180958,0.001018146,0.0003450947,0.000471641,0.00006418966,0.001433625,0.00006982289],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000496788,"about_ca_system_score_gemma":0.001029124,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00104406,"about_ca_topic_score_gemma":0.001587394,"domain_scores_codex":[0.995843,0.0008267007,0.001657833,0.0003874546,0.0005301225,0.00075492],"domain_scores_gemma":[0.9975317,0.00118505,0.0004078009,0.0003572249,0.0003747973,0.0001434325],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.002072423,0.001977598,0.4002123,0.001015462,0.0005007496,0.003702759,0.08168168,0.001510534,0.002533682,0.02215032,0.02505256,0.4575899],"study_design_scores_gemma":[0.000527517,0.00323822,0.9682879,0.0007974299,0.00009675232,0.004560395,0.008754918,0.002316355,0.0007787886,0.003139657,0.007171106,0.0003309197],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9712504,0.00288475,0.00003697238,0.01831041,0.002736434,0.0004363336,0.000003219803,0.0000309391,0.004310518],"genre_scores_gemma":[0.9919933,0.00250275,0.0004749128,0.002206345,0.002481934,0.00003376909,0.00002322347,0.00003265138,0.0002511198],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5680756,"threshold_uncertainty_score":0.8610804,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2005421946447297,"score_gpt":0.4673238457585467,"score_spread":0.266781651113817,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}