{"id":"W4388566534","doi":"10.2196/49459","title":"Strengths and Weaknesses of ChatGPT Models for Scientific Writing About Medical Vitamin B12: Mixed Methods Study","year":2023,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":28,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Transparency (behavior); Strengths and weaknesses; Scientific writing; Inclusion (mineral); Quality (philosophy); Vitamin B12; Data science; Psychology; Medicine; Linguistics; Social psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0114623,0.0001047327,0.0003088943,0.0006865183,0.0005273768,0.00006969144,0.0001643299,0.0001178387,0.00005775564],"category_scores_gemma":[0.004227937,0.00008545559,0.00005379878,0.001377163,0.0004774701,0.0003080291,0.0001661542,0.0003924249,0.0000246409],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007391174,"about_ca_system_score_gemma":0.0005529378,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001574744,"about_ca_topic_score_gemma":0.00007727568,"domain_scores_codex":[0.996936,0.0005312372,0.0006133092,0.0003079676,0.001073152,0.0005383989],"domain_scores_gemma":[0.9927595,0.00534437,0.00008921226,0.0002828045,0.001256212,0.0002679259],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.000181575,0.0006866585,0.001887113,0.001335962,0.00005403296,0.000005076348,0.08478872,0.00001047966,0.001338181,0.001912999,0.006160384,0.9016388],"study_design_scores_gemma":[0.0007854989,0.003937373,0.02347361,0.001490481,0.000044311,0.00002418709,0.6375888,0.2590825,0.0496031,0.01931391,0.004325174,0.0003310091],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9919667,0.0001908816,0.003562008,0.001698634,0.0005805448,0.001651897,0.00001951452,0.00004688068,0.0002829513],"genre_scores_gemma":[0.9974545,0.0001510687,0.001145025,0.00001551373,0.0001576089,0.0006570013,0.00004575718,0.0000172614,0.0003563127],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9013078,"threshold_uncertainty_score":0.5061541,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3206653057813759,"score_gpt":0.6195661621680509,"score_spread":0.298900856386675,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}