{"id":"W4416185268","doi":"10.1016/j.vhri.2025.101539","title":"Evaluating the Performance of Claude 3.7 Sonnet in Data Extraction Automation for Systematic Literature Reviews","year":2025,"lang":"en","type":"article","venue":"Value in Health Regional Issues","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"EVRAZ (Canada)","funders":"","keywords":"Systematic review; Sonnet; Data extraction; Automation; Extraction (chemistry)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04630944,0.001476104,0.002035483,0.01392691,0.001384446,0.004336926,0.001354141,0.002468133,0.004571468],"category_scores_gemma":[0.1715896,0.001115451,0.003012918,0.008892041,0.00074197,0.004011152,0.003499224,0.001011796,0.001656675],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00210114,"about_ca_system_score_gemma":0.009141251,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01111639,"about_ca_topic_score_gemma":0.02128852,"domain_scores_codex":[0.9745064,0.01325563,0.006384274,0.00236411,0.002993556,0.0004959353],"domain_scores_gemma":[0.7296118,0.2373745,0.007769777,0.00679071,0.01671037,0.00174284],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.013114,0.001593614,0.1919102,0.04575509,0.01258804,0.001362751,0.004501302,0.07885695,0.01240839,0.02255803,0.06421328,0.5511383],"study_design_scores_gemma":[0.003910016,0.004421912,0.0655428,0.005603912,0.009499705,0.00205488,0.002612182,0.7706542,0.02328531,0.01963153,0.09234832,0.0004351889],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5848584,0.0284183,0.2373932,0.008917271,0.001399198,0.004056791,0.0486543,0.06823418,0.01806839],"genre_scores_gemma":[0.3964772,0.003568813,0.5573996,0.0007895121,0.0001987649,0.001555119,0.03619536,0.001305718,0.00250997],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9536906,"threshold_uncertainty_score":0.2449107,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1678273020486737,"score_gpt":0.4849505993754512,"score_spread":0.3171232973267775,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}