{"id":"W4405750237","doi":"10.1016/j.jval.2024.10.2413","title":"MSR179 Automated Extraction of Cost-Effectiveness Models Data from Health Technology Assessment Submissions Using Large-Language Models (LLMS): Does the Prompting Approach Matter ?","year":2024,"lang":"en","type":"article","venue":"Value in Health","topic":"Health Systems, Economic Evaluations, Quality of Life","field":"Economics, Econometrics and Finance","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Thermo Fisher Scientific (Canada)","funders":"","keywords":"Computer science; Data extraction; Data science; MEDLINE; Chemistry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.01297217,0.001152781,0.000925567,0.004040426,0.0006198549,0.003454574,0.001245749,0.001236532,0.02384437],"category_scores_gemma":[0.08370708,0.001020496,0.002381093,0.002714398,0.0004118848,0.003335814,0.002010371,0.002235051,0.01832039],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001298746,"about_ca_system_score_gemma":0.004913685,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00355586,"about_ca_topic_score_gemma":0.00532035,"domain_scores_codex":[0.9922115,0.003889398,0.001542212,0.0008698119,0.001267099,0.000219929],"domain_scores_gemma":[0.8890535,0.085855,0.005723329,0.009441071,0.009078861,0.0008482847],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002520634,0.0007481919,0.02642796,0.008754664,0.000880536,0.001351338,0.0016634,0.02611309,0.0212739,0.022266,0.502825,0.3851753],"study_design_scores_gemma":[0.001414877,0.0007250349,0.02362191,0.00241198,0.0006277111,0.0007826692,0.001196322,0.2981389,0.05372515,0.05075173,0.5661331,0.0004706247],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"empirical","genre_scores_codex":[0.07479466,0.001552608,0.2793868,0.008342207,0.001345594,0.00166135,0.4581431,0.1581863,0.01658735],"genre_scores_gemma":[0.2468752,0.0009041156,0.3550832,0.001137394,0.0005449299,0.001520054,0.3775862,0.009057955,0.00729095],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9870278,"threshold_uncertainty_score":0.07976735,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5891400164229476,"score_gpt":0.5241108448239848,"score_spread":0.06502917159896271,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}