{"id":"W4416713244","doi":"10.1016/j.jposna.2025.100294","title":"Not Ready for Prime Time: Limitations of a Retrieval-Augmented Generation Large Language Model in Assessing Risk of Bias in Observational Studies","year":2025,"lang":"en","type":"article","venue":"Journal of the Pediatric Orthopaedic Society of North America","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Observational study; Prime (order theory); Identification (biology); Component (thermodynamics); Risk assessment","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","metaepi_broad"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.7916257,0.004997111,0.01037546,0.01568398,0.003161534,0.0130739,0.008031725,0.004483964,0.00497537],"category_scores_gemma":[0.9185689,0.004480323,0.02133305,0.01720215,0.009297895,0.01324556,0.01204777,0.005109834,0.001115144],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006632487,"about_ca_system_score_gemma":0.01459855,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00616934,"about_ca_topic_score_gemma":0.01239902,"domain_scores_codex":[0.08291784,0.7682253,0.09873766,0.01695014,0.03223891,0.0009301852],"domain_scores_gemma":[0.02309237,0.8954265,0.03845841,0.03144347,0.01099192,0.0005873926],"domain_codex":"methods","domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.01264264,0.0005109359,0.1394584,0.2787853,0.1354014,0.001319214,0.03411784,0.0188251,0.002012957,0.02699399,0.03200221,0.3179299],"study_design_scores_gemma":[0.01858278,0.01039604,0.1103469,0.1780634,0.1389511,0.005109935,0.009934017,0.1821803,0.007368958,0.1932821,0.1411667,0.004617813],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06690497,0.06040676,0.7646103,0.01930797,0.005126584,0.06135472,0.009292952,0.00454412,0.008451739],"genre_scores_gemma":[0.4543827,0.003846577,0.4529216,0.005307456,0.0009460638,0.07880386,0.00208232,0.0009377101,0.0007716398],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9896246,"threshold_uncertainty_score":0.2569626,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3457414614050695,"score_gpt":0.4492989157929659,"score_spread":0.1035574543878964,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}