{"id":"W4414008064","doi":"10.1101/2025.09.03.25334905","title":"INSPECT-SR: a tool for assessing trustworthiness of randomised controlled trials","year":2025,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"Nova Scotia Health Authority; Dalhousie University","funders":"National Institute for Health and Care Research; Queensland University of Technology; Neuroscience Research Australia","keywords":"Trustworthiness; Psychology; Computer science; Social psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","metaepi_broad","scholarly_communication","insufficient_payload"],"consensus_categories":["metaresearch","metaepi_broad"],"category_scores_codex":[0.5501896,0.0007167033,0.03665153,0.001101168,0.0001453022,0.001645641,0.003423379,0.0004286767,0.004365903],"category_scores_gemma":[0.6221925,0.0003036199,0.01908978,0.001100465,0.00008126758,0.0001502066,0.0005221878,0.0003922376,0.00007474253],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004335584,"about_ca_system_score_gemma":0.0007267346,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000007352677,"about_ca_topic_score_gemma":0.0000131153,"domain_scores_codex":[0.8665939,0.06617307,0.05612092,0.002628607,0.007981351,0.0005021678],"domain_scores_gemma":[0.7589158,0.1712026,0.05389254,0.009642042,0.006159432,0.0001875961],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"randomized_trial","study_design_scores_codex":[0.03260837,0.001872535,0.2270698,0.01693865,0.06262596,0.0000517919,0.004612394,0.01388239,0.003291202,0.0339711,0.2823774,0.3206984],"study_design_scores_gemma":[0.3494584,0.000148079,0.02080189,0.005149723,0.04779757,0.000009252782,0.001058724,0.2365279,0.001615014,0.187209,0.1470408,0.003183667],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4440671,0.008904813,0.4977266,0.001311397,0.005077792,0.02540892,0.000340976,0.00002957422,0.01713279],"genre_scores_gemma":[0.9416147,0.0002554653,0.02218878,0.0002715248,0.0006821062,0.004223693,0.00005939592,0.00003665486,0.03066771],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.4975476,"threshold_uncertainty_score":0.9999416,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6793051329169049,"score_gpt":0.5562413940769169,"score_spread":0.123063738839988,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}