{"id":"W4406799061","doi":"10.1007/s13304-025-02106-3","title":"Evaluating human ability to distinguish between ChatGPT-generated and original scientific abstracts","year":2025,"lang":"en","type":"article","venue":"Updates in Surgery","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":6,"is_retracted":false,"has_abstract":false,"ca_institutions":"Providence Health Care; St. Paul's Hospital; University of British Columbia","funders":"","keywords":"Medicine; Surgery; Medical physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.001922847,0.0004610847,0.0003608931,0.0006405071,0.0002303035,0.001296996,0.0003536273,0.0009707661,0.0139617],"category_scores_gemma":[0.02297983,0.0001084302,0.0003311839,0.0003050393,0.0002425018,0.0006216303,0.0004593841,0.0003955311,0.004702981],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002721585,"about_ca_system_score_gemma":0.0003700529,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002292574,"about_ca_topic_score_gemma":0.002309228,"domain_scores_codex":[0.9990253,0.0003256497,0.00007696557,0.0001687107,0.0003482484,0.00005513448],"domain_scores_gemma":[0.9852484,0.01097687,0.0004411481,0.0006438217,0.002301368,0.0003883562],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.008074516,0.001257795,0.06641504,0.002294365,0.0004140266,0.001165463,0.00132171,0.006620197,0.09257669,0.0008098983,0.06524813,0.7538022],"study_design_scores_gemma":[0.001509059,0.006825841,0.5036806,0.0009210923,0.001201327,0.005553606,0.002724854,0.2163706,0.1761598,0.005354023,0.07924392,0.0004554236],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9240029,0.002960696,0.02704516,0.001233888,0.001713893,0.0007529945,0.006092228,0.003931751,0.03226664],"genre_scores_gemma":[0.941882,0.001366847,0.02410401,0.000802512,0.000540267,0.0004342567,0.007372817,0.0003858821,0.02311133],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9980772,"threshold_uncertainty_score":0.04670656,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2768209135401631,"score_gpt":0.510158515235628,"score_spread":0.2333376016954649,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}