{"id":"W4409425199","doi":"10.1101/2025.04.10.25325555","title":"Agreeability testing of AMSTAR-PF, a tool for quality appraisal of systematic reviews of prognostic factor studies","year":2025,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Healthcare Systems and Public Health","field":"Medicine","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Systematic review; Quality (philosophy); Computer science; Biology; MEDLINE; Philosophy; Epistemology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.672354,0.002277722,0.004938381,0.01833303,0.003130166,0.006182841,0.003149343,0.002872897,0.004079647],"category_scores_gemma":[0.8704662,0.0024798,0.01613165,0.01461599,0.00455776,0.008140664,0.01020565,0.004055131,0.0006381743],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00651095,"about_ca_system_score_gemma":0.01063367,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001422803,"about_ca_topic_score_gemma":0.003773144,"domain_scores_codex":[0.1930609,0.4653483,0.2413587,0.0149489,0.08309849,0.002184651],"domain_scores_gemma":[0.06128096,0.7554321,0.06694991,0.03450994,0.08063332,0.00119364],"domain_codex":"methods","domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.01330421,0.000748952,0.1436137,0.1655999,0.05428004,0.0006343278,0.04680669,0.01034713,0.008935302,0.02030852,0.01764928,0.517772],"study_design_scores_gemma":[0.01507783,0.01805364,0.4393228,0.1208075,0.04066759,0.003816359,0.01767572,0.09970402,0.0215533,0.09176654,0.1279182,0.003636549],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2334835,0.03401365,0.538012,0.01018941,0.004661014,0.14513,0.008244927,0.002935177,0.02333026],"genre_scores_gemma":[0.4089488,0.002280038,0.4699642,0.001081297,0.000592247,0.1143069,0.001898537,0.0004190632,0.0005088832],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.327646,"threshold_uncertainty_score":0.4040459,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4143643606593443,"score_gpt":0.5154356560217342,"score_spread":0.1010712953623899,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}