{"id":"W4405775523","doi":"10.2196/65047","title":"Evaluating and Enhancing Japanese Large Language Models for Genetic Counseling Support: Comparative Study of Domain Adaptation and the Development of an Expert-Evaluated Dataset","year":2024,"lang":"en","type":"article","venue":"JMIR Medical Informatics","topic":"BRCA gene mutations in cancer","field":"Biochemistry, Genetics and Molecular Biology","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Preprint; Adaptation (eye); Domain (mathematical analysis); Computer science; Psychology; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01016411,0.001960637,0.0009064919,0.001670075,0.0008475928,0.001284323,0.002399295,0.001757931,0.002612022],"category_scores_gemma":[0.02482514,0.0003999675,0.001386655,0.001205572,0.0007526079,0.001885596,0.00247457,0.002563364,0.001783825],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001751652,"about_ca_system_score_gemma":0.001619681,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01514062,"about_ca_topic_score_gemma":0.02070494,"domain_scores_codex":[0.9924641,0.004837607,0.0004892179,0.001446713,0.0005530121,0.0002094073],"domain_scores_gemma":[0.9773386,0.01597341,0.0005765254,0.002759758,0.002624977,0.0007269459],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.006079213,0.007222815,0.0436078,0.004283823,0.001401502,0.001995007,0.00517267,0.1090618,0.03338514,0.00163267,0.09440979,0.6917477],"study_design_scores_gemma":[0.001834299,0.003713396,0.0598969,0.000510785,0.001329399,0.00108609,0.00458416,0.8175103,0.04484444,0.002950124,0.06130159,0.0004385168],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8900151,0.004279642,0.06199502,0.001491294,0.0007002185,0.002021671,0.01950841,0.01267288,0.007315731],"genre_scores_gemma":[0.7683011,0.0008314705,0.1239653,0.001138521,0.0001467883,0.001563923,0.09898558,0.0004648106,0.004602473],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01514062,"threshold_uncertainty_score":0.05375361,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05644464894886499,"score_gpt":0.408902476028052,"score_spread":0.352457827079187,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}