{"id":"W3158833611","doi":"10.1101/2021.05.04.21256134","title":"Automated Medical Chart Review for Breast Cancer: A Novel Natural Language Processing Software System","year":2021,"lang":"en","type":"preprint","venue":"medRxiv","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Prevention of Organ Failure; University of British Columbia","funders":"","keywords":"Workflow; Pipeline (software); Computer science; Context (archaeology); Software; Chart; Breast cancer; Health care; Artificial intelligence; Data science; Medicine; Cancer; Database; Statistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001945619,0.0006037839,0.0005122508,0.002185942,0.0005347127,0.001533797,0.001132644,0.0006986051,0.004120961],"category_scores_gemma":[0.007082861,0.000387661,0.0006371912,0.001403873,0.000312857,0.001541743,0.001330886,0.0006229948,0.002676162],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009646542,"about_ca_system_score_gemma":0.002904482,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007209518,"about_ca_topic_score_gemma":0.008948483,"domain_scores_codex":[0.9983537,0.0003224757,0.0002235831,0.0005794225,0.000462312,0.00005846404],"domain_scores_gemma":[0.9967981,0.001539641,0.0003070187,0.0004650505,0.0007073801,0.0001827683],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001005293,0.0007339846,0.01797755,0.001250879,0.0002466463,0.001961729,0.001097206,0.008213617,0.07745725,0.005824327,0.1612194,0.7230121],"study_design_scores_gemma":[0.0008076559,0.0004528595,0.03419561,0.0003601755,0.0002692904,0.003825485,0.0007405489,0.6419345,0.1032818,0.0194748,0.1943876,0.0002697453],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.09156505,0.001038848,0.6040384,0.003055988,0.0002524977,0.001546715,0.02919843,0.2626258,0.006678348],"genre_scores_gemma":[0.2019333,0.000493147,0.757808,0.0009198393,0.0001576037,0.0006011216,0.03179922,0.001679424,0.004608278],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007209518,"threshold_uncertainty_score":0.0143351,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01634403388543198,"score_gpt":0.321052070614684,"score_spread":0.304708036729252,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}