{"id":"W4416431165","doi":"10.2196/73822","title":"Identifying Biomedical Entities for Datasets in Scientific Articles: 4-Step Cache-Augmented Generation Approach Using GPT-4o and PubTator 3.0","year":2025,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Biomedical Text Mining and Ontologies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Metadata; Workflow; Data integration; Metadata modeling; Identification (biology)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002012662,0.0001041983,0.0001405104,0.0003905639,0.0003781307,0.000218828,0.0001982543,0.0001574436,0.000004780869],"category_scores_gemma":[0.000368133,0.00009002975,0.00003506224,0.000495612,0.0008157425,0.00002436073,0.0003172651,0.0001778623,0.000001982008],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005788189,"about_ca_system_score_gemma":0.0001683141,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004038736,"about_ca_topic_score_gemma":0.00004456389,"domain_scores_codex":[0.9983959,0.0002337618,0.0002676646,0.0003648974,0.0003143981,0.000423424],"domain_scores_gemma":[0.9994512,0.00007084087,0.0000416478,0.0002244164,0.0001276324,0.00008427633],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002294069,0.0006304198,0.001864145,0.0008744107,0.0001338083,0.000004716516,0.003766499,0.00001022262,0.8583487,0.001131196,0.06010949,0.07289702],"study_design_scores_gemma":[0.008594021,0.001328187,0.00894673,0.0006839797,0.00005508419,0.00004006197,0.04150614,0.3315012,0.4029049,0.002141207,0.2012881,0.001010376],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9338337,0.0009568883,0.06411722,0.0001806985,0.0001649263,0.0004780337,0.0001698962,0.00001146732,0.000087141],"genre_scores_gemma":[0.9923049,0.0000401069,0.005858131,0.00003544932,0.00007112099,0.0001990257,0.001195879,0.000007789774,0.0002875626],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4554438,"threshold_uncertainty_score":0.3671306,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1455260598801369,"score_gpt":0.4481222549381302,"score_spread":0.3025961950579932,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}