{"id":"W4403566032","doi":"10.51731/cjht.2024.1004","title":"Development of an Evaluation Instrument on Artificial Intelligence Search Tools for Evidence Synthesis","year":2024,"lang":"en","type":"article","venue":"Canadian Journal of Health Technologies","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.5763092,0.00198042,0.004783011,0.02356016,0.003327842,0.01185657,0.003294481,0.003630712,0.005554467],"category_scores_gemma":[0.6957069,0.001806806,0.009478489,0.01972358,0.005296044,0.0127344,0.009880504,0.004937913,0.002085153],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01643058,"about_ca_system_score_gemma":0.03932135,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001818945,"about_ca_topic_score_gemma":0.002414621,"domain_scores_codex":[0.3452986,0.422415,0.1641454,0.00684701,0.0569127,0.004381349],"domain_scores_gemma":[0.1608,0.6337789,0.04336783,0.0287646,0.1291138,0.004174896],"domain_codex":"methods","domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.004313425,0.001487221,0.0307722,0.03880391,0.002508323,0.0002564319,0.01236488,0.009487154,0.002885442,0.05821595,0.03133714,0.807568],"study_design_scores_gemma":[0.01255219,0.01678103,0.1388793,0.09038643,0.008969918,0.001276086,0.02327412,0.09878899,0.0403024,0.1754309,0.3906907,0.002668059],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.06468087,0.006690777,0.563463,0.01463127,0.00205722,0.29906,0.006814873,0.003615062,0.03898688],"genre_scores_gemma":[0.06075164,0.001161023,0.7620994,0.001214268,0.0001192043,0.1726341,0.001329146,0.0001665207,0.0005246548],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.4236908,"threshold_uncertainty_score":0.5224862,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.56733717002908,"score_gpt":0.5086916951041844,"score_spread":0.05864547492489558,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}