{"id":"W4403566032","doi":"10.51731/cjht.2024.1004","title":"Development of an Evaluation Instrument on Artificial Intelligence Search Tools for Evidence Synthesis","year":2024,"lang":"en","type":"article","venue":"Canadian Journal of Health Technologies","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004043122,0.00009874898,0.0002725603,0.0008424317,0.000204591,0.00006132351,0.0001900771,0.0001267391,0.00004032075],"category_scores_gemma":[0.004758703,0.00008265393,0.00006057979,0.0003897917,0.000124579,0.0002591167,0.000007188425,0.000292273,0.000009490414],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001247234,"about_ca_system_score_gemma":0.01597014,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001134747,"about_ca_topic_score_gemma":0.002631018,"domain_scores_codex":[0.9980279,0.00007767651,0.001030325,0.0001748689,0.0003592829,0.0003299168],"domain_scores_gemma":[0.9982091,0.000632168,0.0001961696,0.0002036105,0.0006332195,0.0001256887],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00005616948,0.00002808685,0.00006184953,0.000436663,0.00001914771,0.000005212872,0.002296972,0.00007605455,0.0002035597,0.001942123,0.00006009985,0.994814],"study_design_scores_gemma":[0.00003436585,0.004285325,0.001763059,0.01219404,0.0001183404,0.0001622931,0.09844001,0.008585723,0.830724,0.02874041,0.01464941,0.0003029981],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9422035,0.01151492,0.006737446,0.03762536,0.0008612606,0.0009775414,0.000008927394,0.00004863571,0.00002245335],"genre_scores_gemma":[0.9886484,0.0002540679,0.01085691,0.0001174971,0.00004399284,0.00006080425,0.000003750338,0.00001124715,0.000003363094],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9945111,"threshold_uncertainty_score":0.9896084,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.56733717002908,"score_gpt":0.5086916951041844,"score_spread":0.05864547492489558,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}