{"id":"W6887648048","doi":"10.17605/osf.io/498k6","title":"Assessing the research landscape and utility of LLMs in the clinical setting: protocol for a scoping review","year":2023,"lang":"en","type":"other","venue":"Open Science Framework","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Transformer; Natural language; Artificial neural network; Language model; Deep learning; Context (archaeology); Task (project management); Generative grammar","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1032238,0.00386077,0.01131694,0.01928358,0.004298331,0.008930393,0.004854728,0.005705861,0.04118888],"category_scores_gemma":[0.1774343,0.003120818,0.01466328,0.01514549,0.005004304,0.009123676,0.008665243,0.005164182,0.007738108],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02008159,"about_ca_system_score_gemma":0.05776994,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007385313,"about_ca_topic_score_gemma":0.01808064,"domain_scores_codex":[0.9361621,0.02236733,0.02508803,0.004598773,0.00962995,0.00215378],"domain_scores_gemma":[0.888908,0.039263,0.02424916,0.00866114,0.03618946,0.002729271],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002510639,0.0001583888,0.0009916715,0.8829658,0.003004218,0.0004983818,0.001548281,0.0006270937,0.001204715,0.0049443,0.03721349,0.06433299],"study_design_scores_gemma":[0.003351086,0.0005481286,0.002496408,0.8629449,0.005574404,0.0002623012,0.002176327,0.0003833487,0.001072119,0.007094881,0.1139091,0.0001871272],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"protocol","genre_gemma":"protocol","genre_scores_codex":[0.001226723,0.02350276,0.004465211,0.003497641,0.002047532,0.9525617,0.009871954,0.0002510099,0.002575479],"genre_scores_gemma":[0.001818684,0.007515483,0.007769724,0.001432525,0.0001117873,0.9793186,0.001198089,0.00004188491,0.0007932215],"genre_candidate":"protocol","genre_consensus":"protocol","teacher_disagreement_score":0.8967763,"threshold_uncertainty_score":0.5459059,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5550356977208956,"score_gpt":0.6782129478442712,"score_spread":0.1231772501233757,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}