{"id":"W3207783082","doi":"","title":"BI-RADS BERT & Using Section Tokenization to Understand Radiology Reports.","year":2021,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Health Sciences Centre; Sunnybrook Health Science Centre; University of Toronto","funders":"","keywords":"Lexical analysis; Computer science; Artificial intelligence; Lexicon; Natural language processing; Preprocessor; Breast imaging; Section (typography); Radiology; Breast cancer; Mammography; Medicine; Cancer","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001346427,0.0008208808,0.0002431267,0.001228826,0.0002373674,0.001031106,0.0005823156,0.0005222472,0.002818695],"category_scores_gemma":[0.004843607,0.0003448925,0.000875766,0.0007073406,0.0003199613,0.002893032,0.0009201461,0.0009184158,0.003478507],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000729126,"about_ca_system_score_gemma":0.000983939,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005270118,"about_ca_topic_score_gemma":0.007414137,"domain_scores_codex":[0.9993254,0.0002347565,0.000070046,0.0002017946,0.0001172897,0.00005066778],"domain_scores_gemma":[0.9979035,0.0009710671,0.000310965,0.0004386635,0.0003177606,0.00005815968],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009253332,0.0002917449,0.04034976,0.0008330695,0.0002507589,0.0007119549,0.001327512,0.09557635,0.04797072,0.03108102,0.03894969,0.7417319],"study_design_scores_gemma":[0.00002903091,0.0002773481,0.0184955,0.0001312081,0.0001388012,0.001024487,0.0003755736,0.8364249,0.03719014,0.02948964,0.07633642,0.00008700859],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.09954564,0.001333974,0.8691946,0.001151084,0.0004382571,0.0003108378,0.006406951,0.01446342,0.007155191],"genre_scores_gemma":[0.634311,0.001037357,0.3283456,0.0003300045,0.0002532944,0.0003136061,0.02088553,0.001061892,0.01346176],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005270118,"threshold_uncertainty_score":0.01047891,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1166955285620192,"score_gpt":0.2011895285846554,"score_spread":0.0844940000226362,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}