{"id":"W4212984080","doi":"10.2196/preprints.27210","title":"A Question-and-Answer System to Extract Data From Free-Text Oncological Pathology Reports (CancerBERT Network): Development Study (Preprint)","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta; Alberta Health Services","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Terminology; Named-entity recognition; Information retrieval; Pathology; Medicine; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"bench_or_experimental","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"medium","status":"direct model label, unvalidated"}],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002142479,0.0009192619,0.0004849554,0.0009022863,0.000345715,0.0009225119,0.001688474,0.001288598,0.0120415],"category_scores_gemma":[0.005532905,0.0004204582,0.0006376435,0.0005414914,0.0003125732,0.003131193,0.001122717,0.001114106,0.008474491],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001035267,"about_ca_system_score_gemma":0.00153814,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008439544,"about_ca_topic_score_gemma":0.008209541,"domain_scores_codex":[0.9992517,0.0001789816,0.00005411,0.0002774626,0.0001818189,0.00005600719],"domain_scores_gemma":[0.9968612,0.001758496,0.0001133594,0.0002834538,0.0008183693,0.0001650315],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001330186,0.001472049,0.007265956,0.001273552,0.0002330712,0.001044346,0.0009420473,0.02126234,0.04510401,0.005732754,0.1270016,0.787338],"study_design_scores_gemma":[0.0003231461,0.001047411,0.007025729,0.0001330867,0.0001558553,0.0009260415,0.0005607728,0.8328177,0.08235031,0.003188366,0.07135299,0.0001185602],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2394544,0.002021154,0.5069705,0.00252657,0.0009191301,0.002953223,0.02180197,0.2072304,0.01612259],"genre_scores_gemma":[0.3070689,0.0008357311,0.6000243,0.001184101,0.0001655562,0.001564402,0.0564219,0.001835306,0.03089984],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0120415,"threshold_uncertainty_score":0.04028285,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06753405355201075,"score_gpt":0.3209014352907348,"score_spread":0.253367381738724,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}