{"id":"W7117232809","doi":"10.5206/cjils-rcsib.v49i1.23075","title":"Assessing the Effectiveness of AI Tools (Elicit, SciSpace, and Consensus) in Literature Review and Research","year":2025,"lang":"en","type":"article","venue":"Canadian Journal of Information and Library Science","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Automatic summarization; Strengths and weaknesses; Workflow; Relation (database); Systematic review; Sample (material)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","scholarly_communication"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.1099232,0.00007108205,0.0006749879,0.001549587,0.0002643278,0.004142545,0.0006345404,0.00003015031,0.0000716373],"category_scores_gemma":[0.02700641,0.00003133615,0.00007098263,0.004430655,0.0006049686,0.005740987,0.00006440674,0.0002603032,0.000002335494],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001504906,"about_ca_system_score_gemma":0.001584583,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004203646,"about_ca_topic_score_gemma":0.00005772537,"domain_scores_codex":[0.9934958,0.003128124,0.002115462,0.0001267976,0.0009881342,0.000145653],"domain_scores_gemma":[0.9938588,0.003662528,0.0008552429,0.0003769372,0.001017394,0.0002290777],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00002295395,0.00001025213,0.5261464,0.005069639,0.00006039399,0.00002618373,0.005707867,0.0000604625,0.0001614773,0.1684013,0.06124829,0.2330848],"study_design_scores_gemma":[0.0002427009,0.00004365929,0.8816527,0.007464133,0.00003355498,0.0001842738,0.002467589,0.00145478,0.00008320557,0.009579103,0.09670442,0.00008991831],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8655374,0.09153688,0.0005133839,0.02102779,0.0001804152,0.0008288518,0.00001620441,8.674604e-7,0.02035819],"genre_scores_gemma":[0.9957032,0.001688428,0.0004860362,0.001950708,0.000005621764,0.000001751233,5.52271e-7,8.968829e-7,0.0001627816],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3555063,"threshold_uncertainty_score":0.9968913,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3458123296977866,"score_gpt":0.5079584312152706,"score_spread":0.162146101517484,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}