{"id":"W4394696642","doi":"10.2196/57001","title":"Assessing and Optimizing Large Language Models on Spondyloarthritis Multi-Choice Question Answering: Protocol for Enhancement and Assessment","year":2024,"lang":"en","type":"article","venue":"JMIR Research Protocols","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Protocol (science); Computer science; Question answering; Language model; Data science; Artificial intelligence; Natural language processing; Medicine; World Wide Web; Alternative medicine","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.003766692,0.0002201937,0.0002149179,0.0003370265,0.0005278012,0.002403305,0.0004125264,0.0001196925,0.00001667988],"category_scores_gemma":[0.0001826806,0.0001989361,0.00004042952,0.0003502606,0.00007047207,0.001163689,0.0005304385,0.001000241,0.000006123591],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002172323,"about_ca_system_score_gemma":0.0002735079,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006813577,"about_ca_topic_score_gemma":0.00002615519,"domain_scores_codex":[0.9965774,0.000637811,0.0003388574,0.0009263895,0.0007618686,0.0007576293],"domain_scores_gemma":[0.9982616,0.0007200853,0.00006793433,0.0005088213,0.0002122734,0.0002292988],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001873617,0.000990442,0.0008796807,0.01488881,0.00003292095,0.0002017673,0.005278906,0.0004604669,0.01001593,0.1408825,0.001352501,0.8248287],"study_design_scores_gemma":[0.002634314,0.002070315,0.001299532,0.01039847,8.883318e-7,0.0000186469,0.0001418003,0.8238698,0.002057361,0.001422819,0.1557461,0.0003399897],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"protocol","genre_gemma":"protocol","genre_scores_codex":[0.0002128734,0.0000116671,0.3410688,0.001039615,0.00001047569,0.6570324,0.000004630879,0.0002062171,0.0004132983],"genre_scores_gemma":[0.002963984,6.301671e-7,0.1799288,0.00005873567,0.00009957623,0.8166775,0.000003751562,0.00003151313,0.0002354446],"genre_candidate":"protocol","genre_consensus":"protocol","teacher_disagreement_score":0.8244887,"threshold_uncertainty_score":0.9986323,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3059897700266886,"score_gpt":0.628692378855669,"score_spread":0.3227026088289804,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}