{"id":"W4224921946","doi":"10.2196/35475","title":"Using Natural Language Processing and Machine Learning to Preoperatively Predict Lymph Node Metastasis for Non–Small Cell Lung Cancer With Electronic Medical Records: Development and Validation Study","year":2022,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Lung Cancer Diagnosis and Treatment","field":"Medicine","cited_by":27,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Key Research and Development Program of China","keywords":"Receiver operating characteristic; Artificial intelligence; Machine learning; Concordance; Random forest; Computer science; Medicine; Information extraction; Lung cancer; Lymph node; Standardized uptake value; Natural language processing; Text mining; Radiology; Computed tomography; Pathology; Internal medicine","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006586947,0.0002153247,0.0003745417,0.0001227752,0.0003861153,0.00005533058,0.00009228488,0.00005616128,0.0001069091],"category_scores_gemma":[0.00007583629,0.000150052,0.00002601954,0.0002213751,0.0000375362,0.0001385836,0.0001731871,0.0004079981,3.527159e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006395999,"about_ca_system_score_gemma":0.001457818,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001233151,"about_ca_topic_score_gemma":0.0001492759,"domain_scores_codex":[0.9979413,0.00005669592,0.000467452,0.0002116354,0.0009632192,0.0003597462],"domain_scores_gemma":[0.9992208,0.00009124805,0.0001594183,0.00009733503,0.00008526658,0.0003458785],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001261324,0.001661709,0.5206899,0.004402443,0.00113727,0.0001116526,0.1870389,0.0007191187,0.00003963192,0.0000064838,0.0003262053,0.2826054],"study_design_scores_gemma":[0.02338311,0.003794546,0.0291475,0.002180367,0.001618959,0.0003995241,0.05484309,0.8774648,0.002198126,0.000001394313,0.004134464,0.0008341142],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9931408,0.001648479,0.002504437,0.0004480013,0.00005233713,0.002095067,0.00001505804,0.00004529942,0.00005056529],"genre_scores_gemma":[0.9893089,0.00008296403,0.007834619,0.0008161411,0.00006622954,0.001666506,0.00009337239,0.00002754843,0.0001037289],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8767457,"threshold_uncertainty_score":0.6118942,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0173266373551024,"score_gpt":0.3225897557349218,"score_spread":0.3052631183798193,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}