{"id":"W4407009455","doi":"10.1038/s41746-025-01457-w","title":"Language models for data extraction and risk of bias assessment in complementary medicine","year":2025,"lang":"en","type":"article","venue":"npj Digital Medicine","topic":"Complementary and Alternative Medicine Studies","field":"Medicine","cited_by":35,"is_retracted":false,"has_abstract":true,"ca_institutions":"McMaster University; Impact","funders":"Fundamental Research Funds for the Central Universities; China Academy of Chinese Medical Sciences; National Natural Science Foundation of China","keywords":"Computer science; Data extraction; Extraction (chemistry); Data science; Natural language processing; MEDLINE; Chemistry; Chromatography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3983797,0.004714485,0.007704256,0.009756423,0.001412092,0.008753614,0.00453935,0.003885814,0.01202533],"category_scores_gemma":[0.7220018,0.003275139,0.01617535,0.007843927,0.002340341,0.006229269,0.007362201,0.007164626,0.002643055],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005130364,"about_ca_system_score_gemma":0.01443255,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003726914,"about_ca_topic_score_gemma":0.00593061,"domain_scores_codex":[0.4034865,0.5406467,0.03490702,0.008560217,0.01180311,0.0005965452],"domain_scores_gemma":[0.1193084,0.8388985,0.01919546,0.01545956,0.006612069,0.0005259349],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.004551824,0.0003372387,0.01142161,0.1058069,0.0650833,0.0004579034,0.002508961,0.04081935,0.001715955,0.05791183,0.03015879,0.6792264],"study_design_scores_gemma":[0.007572517,0.002419004,0.007263882,0.0421808,0.03347048,0.000804943,0.0006860025,0.3374365,0.007024216,0.4826348,0.07730839,0.001198477],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.002416278,0.01893239,0.9587944,0.004632642,0.0006432906,0.006172272,0.00396028,0.003125117,0.001323358],"genre_scores_gemma":[0.07037468,0.004084911,0.8965622,0.002592992,0.0004930475,0.02298068,0.001953143,0.0004952356,0.0004630611],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6016203,"threshold_uncertainty_score":0.741905,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1603237042729094,"score_gpt":0.4542686677577073,"score_spread":0.2939449634847979,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}