{"id":"W7106820813","doi":"10.48448/sjfy-cz61","title":"Trustworthy Medical Question Answering: An Evaluation-Centric Survey","year":2025,"lang":"","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University; Western University","funders":"","keywords":"Trustworthiness; Software deployment; Adversarial system; Key (lock); Dimension (graph theory); Reliability (semiconductor); Health care","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06002679,0.0007673831,0.0009112404,0.007405187,0.0009875901,0.004084224,0.002281769,0.002012925,0.003218628],"category_scores_gemma":[0.2300116,0.0006376053,0.0009821871,0.005832859,0.002065726,0.007501953,0.003719842,0.002154685,0.001236002],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003712572,"about_ca_system_score_gemma":0.004426266,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00406288,"about_ca_topic_score_gemma":0.003545619,"domain_scores_codex":[0.9475837,0.02647104,0.004711893,0.003464276,0.01647566,0.001293356],"domain_scores_gemma":[0.6726522,0.2687306,0.009648372,0.01245551,0.03395759,0.002555648],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000470526,0.0001878794,0.05132513,0.006906069,0.0003549765,0.0001079737,0.002976601,0.00604005,0.002413483,0.0241149,0.03781836,0.8672841],"study_design_scores_gemma":[0.0002077866,0.001933059,0.1138134,0.02237277,0.001251808,0.003184648,0.007816043,0.119822,0.01934599,0.08196308,0.6278996,0.0003898507],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.1481264,0.5013261,0.2298378,0.04997984,0.001010854,0.001365293,0.006085196,0.004207985,0.05806051],"genre_scores_gemma":[0.7778184,0.1213647,0.07832199,0.008326893,0.001000388,0.000717306,0.008516652,0.001231981,0.002701764],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.9399732,"threshold_uncertainty_score":0.3174558,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09246371048149021,"score_gpt":0.4153009031369939,"score_spread":0.3228371926555036,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}