{"id":"W4416035446","doi":"10.18653/v1/2025.emnlp-main.1398","title":"Trustworthy Medical Question Answering: An Evaluation-Centric Survey","year":2025,"lang":"","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Science Foundation of Shandong Province; Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China; Canadian Institute for Advanced Research","keywords":"Trustworthiness; Natural (archaeology); Empirical research; Questions and answers; Natural language","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.105293,0.0007098584,0.00164463,0.01025449,0.001161036,0.00390107,0.003018315,0.002944615,0.00331608],"category_scores_gemma":[0.3264591,0.0006482564,0.0009799952,0.006909647,0.002666224,0.009119486,0.003853906,0.002078989,0.001243114],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002916307,"about_ca_system_score_gemma":0.00342457,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003518297,"about_ca_topic_score_gemma":0.003982743,"domain_scores_codex":[0.91829,0.04362776,0.006665472,0.004670134,0.02520135,0.001545365],"domain_scores_gemma":[0.4163363,0.5109869,0.0178809,0.01576463,0.03552104,0.00351014],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001083787,0.0003219853,0.1174082,0.007405577,0.0006843519,0.0001904593,0.003932071,0.002132837,0.001436037,0.009884589,0.06363858,0.7918815],"study_design_scores_gemma":[0.000644514,0.002791073,0.2851631,0.02459539,0.00338871,0.006380897,0.01393818,0.07008335,0.01105957,0.06730167,0.5142449,0.0004085782],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.1546699,0.6130073,0.08433972,0.09183656,0.001459087,0.001142871,0.007221902,0.002059544,0.04426315],"genre_scores_gemma":[0.8550937,0.09449416,0.02735151,0.009632669,0.002086706,0.0005210082,0.008508604,0.0004928033,0.001818728],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.105293,"threshold_uncertainty_score":0.5568493,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05583913506053498,"score_gpt":0.3540603383243048,"score_spread":0.2982212032637698,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}