{"id":"W4389814783","doi":"10.1101/2023.12.14.23299971","title":"GPT for RCTs?: Using AI to determine adherence to reporting guidelines","year":2023,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Ottawa Hospital; University of British Columbia","funders":"Canadian Institutes of Health Research","keywords":"Guideline; Medicine; Hyperparameter; Test (biology); Computer science; Artificial intelligence; Machine learning; Medical physics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","scholarly_communication","insufficient_payload"],"consensus_categories":["metaresearch","insufficient_payload"],"category_scores_codex":[0.2135747,0.0006031122,0.006200262,0.0009058904,0.000189972,0.001574446,0.003600998,0.0002247103,0.001298412],"category_scores_gemma":[0.5302734,0.0003266057,0.003041008,0.001808471,0.00002187423,0.00009828951,0.001996942,0.0002801839,0.003183741],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006610859,"about_ca_system_score_gemma":0.0002661918,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001450683,"about_ca_topic_score_gemma":0.0002644722,"domain_scores_codex":[0.9546559,0.002469338,0.03388833,0.002979395,0.005425778,0.0005812976],"domain_scores_gemma":[0.9561812,0.004191665,0.02316725,0.008908931,0.006996047,0.00055493],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000028897,0.00005638062,0.1810609,0.001304679,0.000687078,0.00015943,0.001576149,0.05582068,0.005957389,0.0001763276,0.6226779,0.1304943],"study_design_scores_gemma":[0.0002582066,0.000230786,0.01782134,0.003057596,0.001334076,0.00006222437,0.0006796186,0.3793328,0.001304655,0.02389951,0.5700284,0.001990773],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.5557132,0.0002366325,0.4339179,0.003981755,0.002777188,0.002987445,0.00005578154,0.00003434275,0.0002957673],"genre_scores_gemma":[0.3350427,0.00001871899,0.5861905,0.004992673,0.002035832,0.001395257,0.00003268138,0.0001607369,0.07013096],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.3235121,"threshold_uncertainty_score":0.9999186,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9556734511813036,"score_gpt":0.6544568837914889,"score_spread":0.3012165673898147,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}