{"id":"W4411806137","doi":"10.2196/70706","title":"Extracting Knowledge From Scientific Texts on Patient-Derived Cancer Models Using Large Language Models: Algorithm Development and Validation Study","year":2025,"lang":"en","type":"article","venue":"JMIR Bioinformatics and Biotechnology","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"U.S. National Library of Medicine; National Cancer Institute","keywords":"Computer science; Natural language processing; Algorithm; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002336482,0.000165511,0.0001876403,0.000392937,0.00039021,0.0002763131,0.0002903935,0.0001837712,0.00000116151],"category_scores_gemma":[0.0000110022,0.0001450198,0.0000167301,0.0002993131,0.00005240866,0.000547992,0.000626757,0.0001967733,0.000001600651],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000756029,"about_ca_system_score_gemma":0.0001113888,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006498699,"about_ca_topic_score_gemma":0.00005603814,"domain_scores_codex":[0.9988163,0.00002328971,0.0003895333,0.0003497624,0.0001488607,0.0002723134],"domain_scores_gemma":[0.9993337,0.00003703058,0.0001475845,0.0003803811,0.00005969193,0.00004164733],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000003139486,0.0001593254,0.00005363776,0.00003431248,0.00003356662,0.000002760235,0.02227035,0.0009274894,0.001957608,0.003975377,0.0000108148,0.9705716],"study_design_scores_gemma":[0.000434715,0.00003914725,0.00003195947,0.00008998797,0.000007978556,0.000001926894,0.002760932,0.9780207,0.0166188,0.001616056,0.0002199609,0.0001578574],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5312383,0.0002636173,0.4678264,0.00008228997,0.0001506764,0.0002808244,0.000007227822,0.00008876527,0.00006189963],"genre_scores_gemma":[0.8033307,0.00001806078,0.1965201,0.00005679747,0.000008739961,0.00003267036,0.000006533192,0.000004860303,0.00002156225],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9770932,"threshold_uncertainty_score":0.5913737,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03245524288182356,"score_gpt":0.2980792810151806,"score_spread":0.2656240381333571,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}