{"id":"W4388540984","doi":"10.1136/jme-2023-109549","title":"Exploring the potential utility of AI large language models for medical ethics: an expert panel evaluation of GPT-4","year":2023,"lang":"en","type":"article","venue":"Journal of Medical Ethics","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":38,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta; London Health Sciences Centre; Alberta Health Services; University Health Network; University of Toronto; Dalhousie University; Ontario Shores Centre for Mental Health Sciences; The Scarborough Hospital","funders":"","keywords":"CLARITY; Readability; Counterintuitive; Beneficence; Psychology; Medical ethics; Intraclass correlation; Reliability (semiconductor); Medical education; Engineering ethics; Applied psychology; Social psychology; Medicine; Computer science; Psychometrics; Clinical psychology; Epistemology; Law; Political science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","research_integrity"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.06628073,0.0001020653,0.000384089,0.0001561726,0.0001533014,0.000009353404,0.0003455995,0.000940501,0.0004044563],"category_scores_gemma":[0.06860916,0.0000663928,0.0001977084,0.0002897059,0.00031074,0.0002301154,0.00005349183,0.004172462,0.00000272972],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005836569,"about_ca_system_score_gemma":0.009618257,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007450235,"about_ca_topic_score_gemma":0.0004658201,"domain_scores_codex":[0.9887504,0.001048487,0.001309222,0.0001518027,0.008467706,0.000272359],"domain_scores_gemma":[0.9902283,0.004457674,0.0004439196,0.0003058636,0.004105397,0.0004588287],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001569667,0.001754317,0.0009047507,0.002735012,0.0003275722,0.00009059583,0.4159815,0.0009508086,0.001643802,0.009262208,0.005123073,0.5596567],"study_design_scores_gemma":[0.001135315,0.001222162,0.00263436,0.003121846,0.0004197948,0.0001719973,0.1339904,0.8059658,0.0105329,0.03901397,0.00161793,0.0001735517],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8024108,0.0008283113,0.03528105,0.1590218,0.002030909,0.0003561672,0.000009976607,0.00001741858,0.00004358651],"genre_scores_gemma":[0.9911606,0.003659729,0.0002743024,0.003217728,0.001622171,0.00002339919,0.0000189044,0.00001404356,0.00000910427],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.805015,"threshold_uncertainty_score":0.998125,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7501689367813249,"score_gpt":0.592878808201445,"score_spread":0.1572901285798799,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}