{"id":"W4400694678","doi":"10.2196/55577","title":"Benchmarking Large Language Models for Cervical Spondylosis","year":2024,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Cervical and Thoracic Myelopathy","field":"Medicine","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Peking Union Medical College Hospital; Chinese Academy of Medical Sciences","keywords":"Preprint; Benchmarking; Cervical spondylosis; Computer science; Medicine; World Wide Web; Management; Economics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008157494,0.001551853,0.0008248782,0.00281334,0.0006827534,0.002108137,0.001493642,0.001511745,0.002567967],"category_scores_gemma":[0.03049686,0.0003748689,0.001470907,0.001655109,0.000601003,0.002112472,0.001683012,0.001354405,0.001180797],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002240262,"about_ca_system_score_gemma":0.001883581,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01663377,"about_ca_topic_score_gemma":0.01772325,"domain_scores_codex":[0.9935541,0.003863936,0.0007574778,0.0008716696,0.0006955749,0.0002573055],"domain_scores_gemma":[0.9817914,0.01434291,0.0005272194,0.001182142,0.001748166,0.0004081532],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002359838,0.002050206,0.04713494,0.001350762,0.001217824,0.0007679231,0.001302653,0.5409436,0.006929976,0.002970782,0.02257505,0.3703965],"study_design_scores_gemma":[0.0000877963,0.0005471394,0.007222043,0.00006940584,0.0001366865,0.0001361731,0.0004327046,0.9832214,0.00262453,0.002697412,0.002779867,0.00004480336],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8700562,0.004659958,0.09399568,0.001925027,0.0006020162,0.0008980432,0.01155862,0.009153928,0.00715063],"genre_scores_gemma":[0.9257471,0.0006014281,0.05538711,0.000231224,0.00009066179,0.0003979887,0.01603771,0.0002259181,0.001280848],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01663377,"threshold_uncertainty_score":0.04314148,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06463535175526215,"score_gpt":0.4411146714223853,"score_spread":0.3764793196671232,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}