{"id":"W4415788501","doi":"10.1186/s12891-025-09227-1","title":"Evaluating the performance of five large language models in answering Delphi consensus questions relating to patellar instability and medial patellofemoral ligament reconstruction","year":2025,"lang":"en","type":"article","venue":"BMC Musculoskeletal Disorders","topic":"Total Knee Arthroplasty Outcomes","field":"Medicine","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Manitoba; University of Calgary; University of Ottawa; McMaster University","funders":"Göteborgs Universitet","keywords":"Medial patellofemoral ligament; Sports medicine; Delphi method; Orthopedic surgery; Rehabilitation; Patellofemoral pain syndrome; Multidisciplinary approach; Delphi","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.09568534,0.00126727,0.000902107,0.003377457,0.001616608,0.002408677,0.001468718,0.001657854,0.002424635],"category_scores_gemma":[0.2393794,0.0006564438,0.001425119,0.001471884,0.001810312,0.002167146,0.005139031,0.001232241,0.0007078194],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002728441,"about_ca_system_score_gemma":0.002574294,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002479749,"about_ca_topic_score_gemma":0.004041184,"domain_scores_codex":[0.9049179,0.0778558,0.006189253,0.003104889,0.005955931,0.001976095],"domain_scores_gemma":[0.6864003,0.2706766,0.01162901,0.006472966,0.02209775,0.00272342],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.02113676,0.006001633,0.4184072,0.004430912,0.001287851,0.0009859244,0.1614906,0.03217242,0.01774158,0.003456144,0.004341656,0.3285473],"study_design_scores_gemma":[0.00205617,0.02755032,0.413316,0.001815013,0.001332694,0.001299087,0.1267272,0.3772639,0.02968667,0.008552487,0.009396233,0.001004061],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9871547,0.00007439803,0.00803516,0.000233932,0.00003773585,0.001266982,0.0001415424,0.00009491001,0.002960515],"genre_scores_gemma":[0.973015,0.00006827966,0.02345658,0.0001268173,0.00002267393,0.002322124,0.0003241839,0.00002683286,0.000637566],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9043146,"threshold_uncertainty_score":0.5060385,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02078384556795142,"score_gpt":0.3217574911141608,"score_spread":0.3009736455462094,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}