{"id":"W4364378939","doi":"10.2196/46599","title":"Trialling a Large Language Model (ChatGPT) in General Practice With the Applied Knowledge Test: Observational Study Demonstrating Opportunities and Limitations in Primary Care","year":2023,"lang":"en","type":"article","venue":"JMIR Medical Education","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":192,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Generative grammar; Observational study; Transformer; Computer science; Test (biology); Knowledge management; Artificial intelligence; Medicine; Engineering; Pathology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0409739,0.0006507207,0.001487243,0.001319435,0.001649513,0.002275088,0.001942809,0.002323276,0.001587249],"category_scores_gemma":[0.2092729,0.0009325098,0.0009778454,0.001304615,0.002206891,0.002768203,0.004147073,0.002429422,0.0008110189],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002757997,"about_ca_system_score_gemma":0.003541063,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008304222,"about_ca_topic_score_gemma":0.01260377,"domain_scores_codex":[0.9590678,0.02612635,0.004044669,0.003577475,0.005732357,0.001451329],"domain_scores_gemma":[0.7747363,0.1602768,0.02669408,0.01529277,0.01661229,0.00638775],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001221333,0.004255355,0.7803748,0.001024794,0.0002386771,0.001917809,0.1466763,0.0007175227,0.00116624,0.0002264835,0.002112412,0.06006829],"study_design_scores_gemma":[0.0003750136,0.01741816,0.9028482,0.000841112,0.0002541237,0.002324187,0.05740061,0.00787034,0.001775409,0.001118744,0.0075349,0.0002392383],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9984381,0.0001433996,0.0005206594,0.0001358511,0.000007249896,0.0002639112,0.0000877013,0.00002058221,0.0003827002],"genre_scores_gemma":[0.9971077,0.0001476879,0.001658569,0.0002184777,0.0000167069,0.0004167219,0.0001357097,0.00001893274,0.0002794916],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0409739,"threshold_uncertainty_score":0.2166933,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3570125126813295,"score_gpt":0.4801588478092852,"score_spread":0.1231463351279557,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}