{"id":"W4403566031","doi":"10.51731/cjht.2024.1005","title":"Development of an Evaluation Instrument on Artificial Intelligence Search Tools for Evidence Synthesis","year":2024,"lang":"en","type":"article","venue":"Canadian Journal of Health Technologies","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Systems engineering; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3872665,0.001674808,0.004069077,0.01977365,0.002449028,0.007849774,0.002314939,0.002182369,0.006375382],"category_scores_gemma":[0.5508803,0.00142149,0.008513174,0.01696131,0.002675686,0.008859669,0.007270582,0.003887557,0.002222073],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01211311,"about_ca_system_score_gemma":0.03219163,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00399323,"about_ca_topic_score_gemma":0.00519337,"domain_scores_codex":[0.5604671,0.2624687,0.113209,0.004974268,0.05477576,0.004105228],"domain_scores_gemma":[0.2599242,0.5348212,0.04266886,0.02167432,0.1356495,0.005261911],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00636862,0.002324274,0.05202011,0.02319537,0.003349966,0.0002015608,0.005189926,0.009493231,0.002406463,0.02781526,0.06323373,0.8044015],"study_design_scores_gemma":[0.01689979,0.01785131,0.272439,0.04863397,0.01139407,0.001014061,0.0107279,0.08347107,0.03705802,0.0709535,0.4278294,0.00172788],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.08534002,0.007830835,0.4644356,0.01403153,0.002060837,0.3190297,0.02661948,0.008173621,0.07247844],"genre_scores_gemma":[0.069314,0.001274458,0.7605238,0.001435887,0.0001532938,0.1604126,0.005152912,0.0003600228,0.001373052],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.6127335,"threshold_uncertainty_score":0.7556095,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.56733717002908,"score_gpt":0.5086916951041844,"score_spread":0.05864547492489558,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}