{"id":"W4388540984","doi":"10.1136/jme-2023-109549","title":"Exploring the potential utility of AI large language models for medical ethics: an expert panel evaluation of GPT-4","year":2023,"lang":"en","type":"article","venue":"Journal of Medical Ethics","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":38,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta; London Health Sciences Centre; Alberta Health Services; University Health Network; University of Toronto; Dalhousie University; Ontario Shores Centre for Mental Health Sciences; The Scarborough Hospital","funders":"","keywords":"CLARITY; Readability; Counterintuitive; Beneficence; Psychology; Medical ethics; Intraclass correlation; Reliability (semiconductor); Medical education; Engineering ethics; Applied psychology; Social psychology; Medicine; Computer science; Psychometrics; Clinical psychology; Epistemology; Law; Political science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2345748,0.001044805,0.0009771348,0.002058916,0.001561919,0.004241144,0.002848808,0.002336942,0.003615216],"category_scores_gemma":[0.3527194,0.000742969,0.002007382,0.00156505,0.002949354,0.003716045,0.00906661,0.003577507,0.001004241],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004389972,"about_ca_system_score_gemma":0.005868985,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001445671,"about_ca_topic_score_gemma":0.003260512,"domain_scores_codex":[0.7536817,0.2222121,0.008101824,0.003464204,0.01057774,0.001962413],"domain_scores_gemma":[0.4445271,0.4622377,0.0189186,0.02679366,0.04232732,0.005195675],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.006541527,0.007431778,0.207017,0.006147339,0.0007527148,0.002013847,0.2484375,0.0377579,0.01361444,0.01201312,0.0147282,0.4435446],"study_design_scores_gemma":[0.006688829,0.03190715,0.2521693,0.008257753,0.001477636,0.00353535,0.1507157,0.3093552,0.02845107,0.05325733,0.1521836,0.002001093],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9158415,0.0002918198,0.05372193,0.00373978,0.0001747598,0.009596571,0.0006783719,0.0004040968,0.01555111],"genre_scores_gemma":[0.8380904,0.0003449903,0.1428822,0.001501543,0.0001100001,0.01395666,0.001001077,0.0001439257,0.001969101],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7654252,"threshold_uncertainty_score":0.9439056,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7501689367813249,"score_gpt":0.592878808201445,"score_spread":0.1572901285798799,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}