{"id":"W4411806137","doi":"10.2196/70706","title":"Extracting Knowledge From Scientific Texts on Patient-Derived Cancer Models Using Large Language Models: Algorithm Development and Validation Study","year":2025,"lang":"en","type":"article","venue":"JMIR Bioinformatics and Biotechnology","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"U.S. National Library of Medicine; National Cancer Institute","keywords":"Computer science; Natural language processing; Algorithm; Artificial intelligence","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007430856,0.001507433,0.0009337389,0.002439474,0.0005702385,0.001829807,0.00168934,0.001985969,0.003814998],"category_scores_gemma":[0.03129371,0.000508993,0.00144782,0.001793189,0.0004998115,0.002185933,0.001768413,0.002605102,0.001976585],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001584398,"about_ca_system_score_gemma":0.002943424,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00703259,"about_ca_topic_score_gemma":0.009546185,"domain_scores_codex":[0.9968454,0.001576648,0.0002977126,0.0007578235,0.0003791351,0.0001433251],"domain_scores_gemma":[0.9665914,0.02890348,0.0007239589,0.001432565,0.002043548,0.0003050184],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000894425,0.0007160304,0.0162894,0.0009610422,0.0004409403,0.0006289267,0.0004917585,0.3047446,0.00480359,0.003428443,0.01668393,0.6499171],"study_design_scores_gemma":[0.00008632615,0.00008316986,0.001147202,0.00004747782,0.00004900629,0.0001247065,0.0001039742,0.9921159,0.002067307,0.002271523,0.001889962,0.00001338913],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.2112815,0.003045514,0.7444595,0.002067585,0.0002293704,0.000990857,0.005062273,0.03015935,0.002703941],"genre_scores_gemma":[0.399368,0.0008141652,0.5838881,0.0004633982,0.0001042594,0.0006965506,0.01272289,0.0004077326,0.001534961],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007430856,"threshold_uncertainty_score":0.03929859,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03245524288182356,"score_gpt":0.2980792810151806,"score_spread":0.2656240381333571,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}