{"id":"W4403349562","doi":"10.1016/j.bas.2024.103579","title":"Large Language Models Can Extract Tabular Data from Unstructured Patient Records To Be Used in Large-Scale Retrospective Studies","year":2024,"lang":"en","type":"article","venue":"Brain and Spine","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Scale (ratio); Unstructured data; Computer science; Natural language processing; Data science; Data mining; Geography; Cartography; Big data","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.004797056,0.001706685,0.0009908831,0.00412463,0.0005251322,0.003102215,0.00137806,0.001469453,0.005067323],"category_scores_gemma":[0.03955099,0.000723073,0.00194815,0.003514127,0.0004829626,0.003163455,0.002043006,0.002126236,0.005193648],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007593174,"about_ca_system_score_gemma":0.002705136,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004587412,"about_ca_topic_score_gemma":0.01023296,"domain_scores_codex":[0.9979209,0.0009060891,0.0003702129,0.0004109737,0.0003013155,0.00009048668],"domain_scores_gemma":[0.9661341,0.02598086,0.002401797,0.003272615,0.001828885,0.0003816586],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002780528,0.001446851,0.09888196,0.003558333,0.001751133,0.003108218,0.001622768,0.1089975,0.01815877,0.01434431,0.08932113,0.6560286],"study_design_scores_gemma":[0.0003200526,0.0005253029,0.01385577,0.0008778032,0.0008203308,0.001227439,0.001445596,0.831727,0.011088,0.09530429,0.04257283,0.0002355596],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04322971,0.002128405,0.8314976,0.002855741,0.0004998421,0.001173604,0.09365861,0.0223482,0.002608346],"genre_scores_gemma":[0.3305276,0.001987288,0.529466,0.001699673,0.0004590196,0.001903356,0.1308135,0.00120778,0.001935739],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.995203,"threshold_uncertainty_score":0.02536958,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03842732420521019,"score_gpt":0.337877959462927,"score_spread":0.2994506352577169,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}