{"id":"W4280505624","doi":"10.1186/s12874-022-01583-z","title":"Automated medical chart review for breast cancer outcomes research: a novel natural language processing extraction system","year":2022,"lang":"en","type":"article","venue":"BMC Medical Research Methodology","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":23,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia Hospital; Prevention of Organ Failure; University of British Columbia","funders":"University of British Columbia","keywords":"Computer science; Data extraction; Pipeline (software); Artificial intelligence; Medicine; Test (biology); Breast cancer; Medical physics; Natural language processing; Machine learning; Cancer; MEDLINE","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.05419556,0.0002166733,0.0006690221,0.0002821903,0.0007818582,0.00002752378,0.001175662,0.0005327071,0.0009034128],"category_scores_gemma":[0.07291961,0.0001633387,0.0002026545,0.0007232598,0.001103508,0.000006599272,0.001109228,0.001967151,0.00001117487],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002172322,"about_ca_system_score_gemma":0.00365308,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004138908,"about_ca_topic_score_gemma":0.0004304516,"domain_scores_codex":[0.9837592,0.009799409,0.0006484942,0.0008848476,0.003655715,0.001252389],"domain_scores_gemma":[0.9906528,0.007421293,0.0001386283,0.0004958069,0.0006503983,0.0006410949],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001234911,0.0004877238,0.0009320667,0.007162559,0.0002094943,0.000170855,0.0003849047,0.000004024013,0.03962294,0.0003949715,0.09462986,0.8547657],"study_design_scores_gemma":[0.01537996,0.004648976,0.0240705,0.01173842,0.0002703758,0.0106432,0.03656336,0.137651,0.007555533,0.0002063831,0.7488115,0.002460786],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.1619209,0.5129563,0.1892819,0.1175183,0.006783263,0.007211662,0.001257714,0.001684196,0.001385746],"genre_scores_gemma":[0.8981782,0.01020161,0.06832094,0.005922023,0.003268315,0.008468836,0.0006932747,0.0001904082,0.004756371],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8523049,"threshold_uncertainty_score":0.9891736,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4498042970878462,"score_gpt":0.6128187548373685,"score_spread":0.1630144577495222,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}