{"id":"W4405189507","doi":"10.2196/55827","title":"Evaluation of RMES, an Automated Software Tool Utilizing AI, for Literature Screening with Reference to Published Systematic Reviews as Case-Studies: Development and Usability Study","year":2024,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Systematic review; Usability; MEDLINE; Medicine; Medical literature; Computer science; Workload; Medical physics; Information retrieval; Pathology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2319635,0.002939749,0.003234914,0.02215357,0.001166061,0.004021479,0.002780577,0.001548047,0.005756114],"category_scores_gemma":[0.4958639,0.002451263,0.004477776,0.01051563,0.0009665245,0.005904717,0.005072574,0.001266784,0.001384226],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00210743,"about_ca_system_score_gemma":0.006182642,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002311739,"about_ca_topic_score_gemma":0.004312058,"domain_scores_codex":[0.8411041,0.09641877,0.04299828,0.005105264,0.0135902,0.000783365],"domain_scores_gemma":[0.2379516,0.6779028,0.02319608,0.01966595,0.04033367,0.0009498866],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.007494316,0.001397478,0.0408668,0.03748944,0.006148184,0.0005984823,0.007776069,0.005405701,0.009551743,0.002325477,0.02423472,0.8567116],"study_design_scores_gemma":[0.02214717,0.01761379,0.1880909,0.0339763,0.02713335,0.006116177,0.008753503,0.4939094,0.06298865,0.0115386,0.1254494,0.002282765],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2604132,0.007043163,0.5616198,0.003201762,0.0006307241,0.06333414,0.01613494,0.08005092,0.007571407],"genre_scores_gemma":[0.1285499,0.001281135,0.8286192,0.0004997179,0.0001076902,0.03483826,0.003443919,0.001760333,0.0008998453],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7680365,"threshold_uncertainty_score":0.9471257,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.829700384117312,"score_gpt":0.6546861646467547,"score_spread":0.1750142194705573,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}