{"id":"W4392413980","doi":"10.2196/49997","title":"A Case Demonstration of the Open Health Natural Language Processing Toolkit From the National COVID-19 Cohort Collaborative and the Researching COVID to Enhance Recovery Programs for a Natural Language Processing System for COVID-19 or Postacute Sequelae of SARS CoV-2 Infection: Algorithm Development and Validation","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Center for Advancing Translational Sciences; U.S. National Library of Medicine; National Heart, Lung, and Blood Institute","keywords":"Coronavirus disease 2019 (COVID-19); Artificial intelligence; Natural language processing; Severe acute respiratory syndrome coronavirus 2 (SARS-CoV-2); Preprint; Task (project management); 2019-20 coronavirus outbreak; Computer science; Medicine; World Wide Web; Pathology; Disease; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00403316,0.0001483293,0.0002696597,0.00008242183,0.0006748602,0.0007448804,0.000493498,0.00008073633,6.13512e-7],"category_scores_gemma":[0.001843914,0.00007752807,0.00003718403,0.0006160346,0.0001994526,0.0009134133,0.0003722538,0.000276396,1.394519e-7],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004991845,"about_ca_system_score_gemma":0.006617998,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004500952,"about_ca_topic_score_gemma":0.0008670226,"domain_scores_codex":[0.9977387,0.0002294721,0.0007741275,0.0002310092,0.0008075375,0.0002191523],"domain_scores_gemma":[0.9975696,0.001361738,0.0004113413,0.0001691373,0.0003270227,0.000161228],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003205513,0.00003674324,0.00007822945,0.005306698,0.0001251477,0.00001841904,0.2912093,0.0001806823,0.00006965015,0.00161606,0.0007913166,0.7002472],"study_design_scores_gemma":[0.001106607,0.0001062535,0.00002178096,0.0009690438,0.00002193549,0.0004460195,0.0180596,0.9763154,0.0009050728,0.0001713418,0.001729868,0.0001470998],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1315068,0.000892421,0.8547984,0.007280185,0.0001752899,0.005141899,0.0001156935,0.00007080063,0.00001855767],"genre_scores_gemma":[0.8357928,0.00001199974,0.1590448,0.004060905,0.00008139558,0.000892602,0.00008311526,0.00001003767,0.00002234738],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9761347,"threshold_uncertainty_score":0.9990135,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05779017490292085,"score_gpt":0.4253979188147319,"score_spread":0.3676077439118111,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}