{"id":"W4280505624","doi":"10.1186/s12874-022-01583-z","title":"Automated medical chart review for breast cancer outcomes research: a novel natural language processing extraction system","year":2022,"lang":"en","type":"article","venue":"BMC Medical Research Methodology","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":23,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia Hospital; Prevention of Organ Failure; University of British Columbia","funders":"University of British Columbia","keywords":"Computer science; Data extraction; Pipeline (software); Artificial intelligence; Medicine; Test (biology); Breast cancer; Medical physics; Natural language processing; Machine learning; Cancer; MEDLINE","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004661462,0.001418257,0.001090727,0.005670707,0.0007449414,0.002251551,0.00191755,0.001015584,0.004556406],"category_scores_gemma":[0.01719589,0.0006298064,0.00125461,0.003326218,0.0004006961,0.002518205,0.002192393,0.001056964,0.004207815],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001340102,"about_ca_system_score_gemma":0.004365469,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005084616,"about_ca_topic_score_gemma":0.00606077,"domain_scores_codex":[0.9949051,0.001044422,0.001129021,0.001526716,0.001241924,0.0001529106],"domain_scores_gemma":[0.9858915,0.007003255,0.001881223,0.001362139,0.003454845,0.0004069291],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007800992,0.0006575687,0.01402188,0.002536734,0.0002590441,0.001851643,0.001132794,0.005209648,0.03972635,0.002493287,0.09545905,0.8358719],"study_design_scores_gemma":[0.00123828,0.001020281,0.05458353,0.001097029,0.0007266141,0.005084754,0.001348531,0.5814076,0.1336571,0.02303428,0.1962022,0.0005997939],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04603071,0.001503074,0.7411892,0.002120907,0.0003578186,0.003336675,0.05135918,0.1498698,0.004232698],"genre_scores_gemma":[0.07702602,0.0005272246,0.872169,0.0005278724,0.000225447,0.001986339,0.04463043,0.0007878894,0.002119691],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005670707,"threshold_uncertainty_score":0.02465242,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4498042970878462,"score_gpt":0.6128187548373685,"score_spread":0.1630144577495222,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}