{"id":"W4412432271","doi":"10.1016/j.jval.2025.04.1240","title":"MSR88 Key Considerations in the Use of Large Language Models for Data Extraction in Health Economics and Outcomes Research","year":2025,"lang":"en","type":"article","venue":"Value in Health","topic":"Chronic Disease Management Strategies","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"EVERSANA (Canada)","funders":"","keywords":"Key (lock); Data extraction; Data science; Outcomes research; Computer science; Econometrics; Economics; MEDLINE; Medicine; Political science; Alternative medicine; Computer security","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2203149,0.001838291,0.002685142,0.003782761,0.002427124,0.01500808,0.005990577,0.003478649,0.02195206],"category_scores_gemma":[0.6106238,0.002587151,0.004617692,0.007278717,0.003977076,0.01705737,0.007423295,0.009831879,0.01267573],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002175124,"about_ca_system_score_gemma":0.007967002,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006591966,"about_ca_topic_score_gemma":0.01296473,"domain_scores_codex":[0.7123774,0.2496991,0.01803306,0.006843656,0.0120604,0.0009864179],"domain_scores_gemma":[0.2235736,0.7168406,0.006216619,0.03604042,0.01615435,0.00117444],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.004250837,0.0004596178,0.01933261,0.007223543,0.002997573,0.001961808,0.005606404,0.02815341,0.008756035,0.3027932,0.1697011,0.4487639],"study_design_scores_gemma":[0.0008872081,0.0004539363,0.006883066,0.003102861,0.000772284,0.001982129,0.002671644,0.1919854,0.0141445,0.5620371,0.2145915,0.0004884508],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.007027329,0.001812783,0.9206505,0.04908321,0.0008936496,0.0009501621,0.00697618,0.005657722,0.006948383],"genre_scores_gemma":[0.08454822,0.001162697,0.8929503,0.005955761,0.001365042,0.002455442,0.00547182,0.002363297,0.003727413],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7796851,"threshold_uncertainty_score":0.9614905,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4409367267451672,"score_gpt":0.5246048573887429,"score_spread":0.0836681306435757,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}