{"id":"W4412163803","doi":"10.1158/1557-3265.aimachine-b006","title":"Abstract B006: Using large language models for scalable extraction of real-world progression events across multiple cancer types","year":2025,"lang":"en","type":"article","venue":"Clinical Cancer Research","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Cancer; Computer science; Scalability; Extraction (chemistry); Medicine; Internal medicine; Chemistry; Database","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005256624,0.0017622,0.0006679441,0.002300813,0.0007658226,0.002220364,0.001557189,0.001205625,0.005310883],"category_scores_gemma":[0.02126708,0.000617081,0.002672448,0.001418959,0.0005329835,0.002823179,0.002479422,0.002281355,0.004299686],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001102956,"about_ca_system_score_gemma":0.002520679,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01396902,"about_ca_topic_score_gemma":0.02105846,"domain_scores_codex":[0.9971871,0.001168497,0.0003445997,0.0008145488,0.0003649493,0.0001203147],"domain_scores_gemma":[0.989126,0.007554626,0.0006332336,0.001286023,0.001141359,0.0002588455],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002068758,0.000670922,0.04309064,0.002509259,0.0008607274,0.001101602,0.001796528,0.0963306,0.02139459,0.007917749,0.09217531,0.7300833],"study_design_scores_gemma":[0.0002715648,0.0003071756,0.008315811,0.0002187748,0.0002254687,0.0004069629,0.000547138,0.9288609,0.0137214,0.01851787,0.028464,0.0001428678],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.09869745,0.002182539,0.777882,0.003591157,0.0005814385,0.001124695,0.04445817,0.06784624,0.003636273],"genre_scores_gemma":[0.2644066,0.0005414493,0.663772,0.0008340222,0.0001983563,0.0009090314,0.0651534,0.001263786,0.00292123],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01396902,"threshold_uncertainty_score":0.02780002,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2221175994035013,"score_gpt":0.6034192008268803,"score_spread":0.381301601423379,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}