{"id":"W4404771219","doi":"10.1101/2024.11.25.24316905","title":"Assessing the feasibility and impact of clinical trial trustworthiness checks via an application to Cochrane Reviews: Stage 2 of the INSPECT-SR project","year":2024,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University; Centre for Global Health Research","funders":"National Institute for Health and Care Research","keywords":"Trustworthiness; Stage (stratigraphy); Systematic review; Cochrane collaboration; Process management; Computer science; Business; Political science; MEDLINE; Computer security; Law; Geology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.8704975,0.006411104,0.01277885,0.0302758,0.004760027,0.01550151,0.01006489,0.009173124,0.01813856],"category_scores_gemma":[0.9547894,0.01017657,0.02560558,0.02462416,0.008801161,0.01524244,0.02103887,0.01152357,0.004841675],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01516292,"about_ca_system_score_gemma":0.08294906,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003400487,"about_ca_topic_score_gemma":0.005246628,"domain_scores_codex":[0.08983699,0.6910276,0.1560998,0.01160714,0.04935594,0.002072505],"domain_scores_gemma":[0.01531847,0.8079519,0.04981941,0.05695442,0.06826755,0.00168813],"domain_codex":"methods","domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.01884967,0.001294257,0.02228514,0.2899502,0.02262038,0.0006092586,0.02683959,0.00986116,0.003637482,0.02369951,0.04438956,0.5359639],"study_design_scores_gemma":[0.04377498,0.02113044,0.05230909,0.3413542,0.04625204,0.001745892,0.006548982,0.1264587,0.02502157,0.10874,0.2228451,0.003819091],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"protocol","genre_gemma":"empirical","genre_scores_codex":[0.02402379,0.009144321,0.4346521,0.01057427,0.002763874,0.4984657,0.005902605,0.006393294,0.008080048],"genre_scores_gemma":[0.02977402,0.001054375,0.6178016,0.0009917542,0.000285473,0.3478542,0.001108699,0.000626251,0.0005035473],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1295025,"threshold_uncertainty_score":0.1596997,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7921343506849283,"score_gpt":0.657622913223708,"score_spread":0.1345114374612203,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}