{"id":"W4212984080","doi":"10.2196/preprints.27210","title":"A Question-and-Answer System to Extract Data From Free-Text Oncological Pathology Reports (CancerBERT Network): Development Study (Preprint)","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta; Alberta Health Services","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Terminology; Named-entity recognition; Information retrieval; Pathology; Medicine; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"bench_or_experimental","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"medium","status":"direct model label, unvalidated"}],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","open_science"],"consensus_categories":[],"category_scores_codex":[0.003036917,0.0005333874,0.001037697,0.000098956,0.0001943277,0.0006589384,0.003909651,0.0004794252,0.0001557319],"category_scores_gemma":[0.0003781071,0.0004766612,0.00007843938,0.0002101811,0.00003203844,0.0002827878,0.03569802,0.0009253638,0.00003448983],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005128348,"about_ca_system_score_gemma":0.00103294,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002316986,"about_ca_topic_score_gemma":0.004601403,"domain_scores_codex":[0.9923813,0.0007022952,0.001533055,0.004080281,0.0007118256,0.0005912287],"domain_scores_gemma":[0.9903991,0.0003426628,0.0004870206,0.00820111,0.0002327129,0.0003374267],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001026246,0.002725681,0.2101353,0.0004915362,0.001494015,0.03861374,0.01799669,0.236702,0.00009722113,0.008481544,0.01327446,0.4698851],"study_design_scores_gemma":[0.00148764,0.0003789257,0.2705704,0.002846985,0.0002784678,0.001915921,0.002262453,0.694131,0.0001100161,0.005753178,0.01639789,0.003867122],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2326761,0.0005676193,0.7594718,0.0006389307,0.003224673,0.001521998,0.000007185847,0.0005045094,0.001387229],"genre_scores_gemma":[0.5372075,0.00001613369,0.4614646,0.0003122874,0.0004009719,0.0003215724,0.00006732949,0.00001773501,0.0001919274],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.466018,"threshold_uncertainty_score":0.9997685,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06753405355201075,"score_gpt":0.3209014352907348,"score_spread":0.253367381738724,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}