{"id":"W4405623317","doi":"10.2196/67056","title":"Automated Pathologic TN Classification Prediction and Rationale Generation From Lung Cancer Surgical Pathology Reports Using a Large Language Model Fine-Tuned With Chain-of-Thought: Algorithm Development and Validation Study","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Korea Health Industry Development Institute","keywords":"Context (archaeology); Computer science; Artificial intelligence; Medicine; Lung cancer; Natural language processing; Parsing; Medical physics; Pathology; Machine learning","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005212557,0.001321018,0.0009049135,0.002345742,0.0006341191,0.001711678,0.001994506,0.00156346,0.00219378],"category_scores_gemma":[0.01512398,0.0006396582,0.001722797,0.0009447149,0.0005881204,0.001464248,0.001463641,0.001827458,0.001054992],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002010173,"about_ca_system_score_gemma":0.003648643,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01389387,"about_ca_topic_score_gemma":0.01735596,"domain_scores_codex":[0.9979624,0.0006936023,0.0002436784,0.0006654164,0.0002963383,0.0001387075],"domain_scores_gemma":[0.9862634,0.01080997,0.0005916671,0.000729806,0.001360909,0.0002443731],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001239178,0.001351656,0.03543473,0.0005613673,0.0004578583,0.0007436695,0.0005362778,0.2939458,0.01207846,0.002403096,0.009580777,0.6416671],"study_design_scores_gemma":[0.00006884555,0.00007771966,0.001350587,0.00001981939,0.00004096921,0.00008450875,0.00005782606,0.9940118,0.00278338,0.0009461463,0.0005446324,0.00001367651],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.392909,0.001387885,0.5580955,0.0009944729,0.0002206141,0.001643972,0.003003712,0.03970423,0.002040722],"genre_scores_gemma":[0.4976084,0.0002554599,0.492778,0.0003427984,0.00006679737,0.000491222,0.007035918,0.0003103032,0.001111056],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01389387,"threshold_uncertainty_score":0.02762598,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0325633751547172,"score_gpt":0.3454929506506829,"score_spread":0.3129295754959657,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}