{"id":"W4411199189","doi":"10.2196/76773","title":"Toward Cross-Hospital Deployment of Natural Language Processing Systems: Model Development and Validation of Fine-Tuned Large Language Models for Disease Name Recognition in Japanese","year":2025,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Software deployment; Computer science; Robustness (evolution); Artificial intelligence; Natural language processing; Medicine; World Wide Web; Software engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007275371,0.001928497,0.0009292287,0.001125512,0.0007051582,0.001276303,0.00193937,0.001407893,0.001232968],"category_scores_gemma":[0.01443177,0.0007592635,0.001528885,0.0007235297,0.0006645106,0.001944709,0.002098267,0.002228978,0.000798408],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002251429,"about_ca_system_score_gemma":0.002279799,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.04932709,"about_ca_topic_score_gemma":0.03698658,"domain_scores_codex":[0.9977741,0.0008589361,0.0002061039,0.0008037845,0.000164652,0.000192483],"domain_scores_gemma":[0.9933381,0.004002125,0.0003711146,0.0006867602,0.00132337,0.0002784849],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001412926,0.001275228,0.03359226,0.0004943704,0.000831371,0.0005491719,0.0009618633,0.7382925,0.01423662,0.0009754436,0.006025416,0.2013529],"study_design_scores_gemma":[0.00005209686,0.0002021284,0.00407044,0.00001803547,0.00009402445,0.00004175411,0.0001297391,0.9905104,0.003927264,0.0003846358,0.000538462,0.00003097143],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8958631,0.00168812,0.09335351,0.0006953171,0.0002947397,0.0005926345,0.001590697,0.003787571,0.002134277],"genre_scores_gemma":[0.9405956,0.0003124083,0.05248284,0.0002640691,0.00005722976,0.0004833482,0.004325243,0.0001784099,0.001300869],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04932709,"threshold_uncertainty_score":0.09807998,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02115195569370589,"score_gpt":0.3070880230549075,"score_spread":0.2859360673612016,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}