{"id":"W4391754926","doi":"10.1101/2024.02.09.24302591","title":"Will ChatGPT-4 improve the quality of medical abstracts?","year":2024,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Montreal Children's Hospital; Université de Montréal; Centre Hospitalier Universitaire Sainte-Justine","funders":"","keywords":"Quality (philosophy); Quality Score; Mean difference; Significant difference; Minor (academic); Medical physics; Computer science; Medical education; Psychology; Medicine; Statistics; Mathematics; Operations management; Engineering; Political science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3028109,0.001334609,0.00298137,0.005266272,0.001968721,0.008288901,0.003282492,0.00274863,0.01480016],"category_scores_gemma":[0.6742436,0.001235781,0.005411017,0.004246085,0.002802444,0.006262943,0.00502838,0.003458866,0.003491151],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005801889,"about_ca_system_score_gemma":0.01080151,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008429298,"about_ca_topic_score_gemma":0.001260292,"domain_scores_codex":[0.6563153,0.2393963,0.05089734,0.008129597,0.04248422,0.00277727],"domain_scores_gemma":[0.1443636,0.6198599,0.1090743,0.03486885,0.08233958,0.0094937],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.03529288,0.002604084,0.04032901,0.08391064,0.003780544,0.0006380632,0.01760218,0.001293948,0.01057688,0.001813631,0.06449855,0.7376596],"study_design_scores_gemma":[0.05311795,0.0927432,0.4168728,0.07365807,0.01698887,0.007992689,0.01808634,0.01378383,0.04402245,0.0252196,0.234354,0.003160166],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6887949,0.06610969,0.06524014,0.08107755,0.01584266,0.04045701,0.008787587,0.009004428,0.024686],"genre_scores_gemma":[0.7355086,0.01554933,0.1679868,0.01381421,0.006623855,0.05139352,0.002981366,0.001505043,0.004637237],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6971891,"threshold_uncertainty_score":0.8597583,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1844758533670177,"score_gpt":0.4880808714032643,"score_spread":0.3036050180362466,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}