{"id":"W4401752756","doi":"10.1109/ichi61247.2024.00089","title":"Seeing Beyond Borders: Evaluating LLMs in Multilingual Ophthalmological Question Answering","year":2024,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Question answering; Computer science; Natural language processing; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009937399,0.00187777,0.0009147196,0.002721221,0.0008969547,0.003399943,0.001841508,0.003566447,0.00638207],"category_scores_gemma":[0.04475634,0.0003409513,0.001396583,0.001397496,0.0007918563,0.005946371,0.004552437,0.002216505,0.002945817],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001681129,"about_ca_system_score_gemma":0.001799696,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009454943,"about_ca_topic_score_gemma":0.01048307,"domain_scores_codex":[0.9914892,0.004930401,0.0008285678,0.001652803,0.0008322559,0.0002669388],"domain_scores_gemma":[0.9666021,0.02896911,0.0007238337,0.001160007,0.001466199,0.00107882],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0125756,0.00429309,0.04818094,0.006700525,0.001997447,0.001045249,0.006390338,0.08669194,0.01663871,0.00423754,0.05182247,0.7594261],"study_design_scores_gemma":[0.001259083,0.00401482,0.02517669,0.0005702294,0.0009407457,0.0008416344,0.005976605,0.907119,0.01241116,0.01676813,0.02466086,0.0002609993],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7970575,0.0139126,0.127336,0.003393289,0.001087326,0.002496629,0.01466046,0.02248028,0.01757578],"genre_scores_gemma":[0.8659086,0.0009420548,0.1021271,0.0009876916,0.0003338974,0.0008831946,0.02664955,0.0003261898,0.001841768],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009937399,"threshold_uncertainty_score":0.05255467,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0550715705987934,"score_gpt":0.3926561732002667,"score_spread":0.3375846026014733,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}