{"id":"W4411199189","doi":"10.2196/76773","title":"Toward Cross-Hospital Deployment of Natural Language Processing Systems: Model Development and Validation of Fine-Tuned Large Language Models for Disease Name Recognition in Japanese","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Software deployment; Computer science; Robustness (evolution); Artificial intelligence; Natural language processing; Medicine; World Wide Web; Software engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004756143,0.0001063603,0.0002106847,0.0001385602,0.00003455097,0.00005605023,0.0002674427,0.00008105272,9.249582e-7],"category_scores_gemma":[0.0001231209,0.0000920091,0.00003054678,0.0001514508,0.00003230185,0.0006658286,0.0001751813,0.0001084656,3.637613e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000059279,"about_ca_system_score_gemma":0.0002967839,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001108622,"about_ca_topic_score_gemma":0.000004136693,"domain_scores_codex":[0.9985096,0.00001734274,0.0007396389,0.0001175657,0.0004297975,0.0001860437],"domain_scores_gemma":[0.9993561,0.00005714181,0.0002052818,0.0001655732,0.0001279472,0.00008795599],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001630424,0.0007435169,0.0009140082,0.01989411,0.00007366754,0.00001706883,0.5443229,0.02762292,0.0001159051,0.006086001,0.00008405471,0.3999628],"study_design_scores_gemma":[0.0009150556,0.000014953,0.00006018844,0.0007380822,0.000005716132,8.343939e-7,0.00301859,0.9940007,0.0008493424,0.0002960557,0.000004409492,0.00009604076],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5684536,0.0003178365,0.4306749,0.00006348905,0.00006007217,0.0003567683,0.00000954997,0.00002682978,0.00003691586],"genre_scores_gemma":[0.9587466,0.000004196101,0.04095299,0.0000840135,0.00001029987,0.0001217704,0.00004385907,0.000004343608,0.00003195473],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9663778,"threshold_uncertainty_score":0.3752022,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02115195569370589,"score_gpt":0.3070880230549075,"score_spread":0.2859360673612016,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}