{"id":"W4417457716","doi":"10.1177/21501319251404193","title":"Balancing Model Complexity and Clinical Deployability in Deep Learning for Sociodemographic Information Extraction","year":2025,"lang":"en","type":"article","venue":"Journal of Primary Care & Community Health","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"North York General Hospital; University Health Network; University of Toronto","funders":"Canadian Institutes of Health Research","keywords":"Documentation; Deep learning; Convolutional neural network; Information overload; Artificial neural network; Binary classification; Health care; Unstructured data; Medical record; Data extraction","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005549239,0.00009443836,0.0003761506,0.0002686762,0.0004748703,0.00008295494,0.0004587611,0.00009278757,2.165261e-7],"category_scores_gemma":[0.0003910162,0.00009577027,0.0001162922,0.0002147408,0.00005841404,0.001198317,0.0002253871,0.001610419,1.055922e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001005484,"about_ca_system_score_gemma":0.001089488,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000661285,"about_ca_topic_score_gemma":0.002003548,"domain_scores_codex":[0.9974915,0.0009729227,0.001051258,0.00009577359,0.000189985,0.0001985617],"domain_scores_gemma":[0.9978102,0.0008212809,0.0005772468,0.0003398607,0.0003544477,0.00009699126],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00007960168,0.000138005,0.2423843,0.0009775342,0.0000209377,4.713473e-7,0.007400397,0.01985234,0.000008545028,0.003135293,0.00003832114,0.7259643],"study_design_scores_gemma":[0.0009479509,0.0002119797,0.428715,0.0001853167,0.000007539413,0.000009742874,0.001235524,0.557731,0.000001461165,0.01070929,0.0001714369,0.00007370239],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3744885,0.0004060504,0.6236966,0.0009070114,0.0001575333,0.000146334,7.129996e-7,0.0000204974,0.0001767404],"genre_scores_gemma":[0.8580216,0.0002892134,0.1405531,0.001105006,0.00001987178,0.000002163748,0.00000529724,0.000002759295,9.630043e-7],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7258906,"threshold_uncertainty_score":0.6996558,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06077800822176373,"score_gpt":0.3738910076647382,"score_spread":0.3131129994429745,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}