{"id":"W4415297078","doi":"10.1016/j.mlwa.2025.100758","title":"Prompt design for medical question answering with Large Language Models","year":2025,"lang":"en","type":"article","venue":"Machine Learning with Applications","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Variety (cybernetics); Workflow; Language model; Tree (set theory); Simple (philosophy); Sonnet","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005158338,0.00136943,0.000948911,0.001004489,0.0004537645,0.002016508,0.002338891,0.001746652,0.00774077],"category_scores_gemma":[0.02138286,0.000539762,0.001616481,0.0007880782,0.0005548566,0.004284337,0.00231139,0.002790107,0.003769831],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001399024,"about_ca_system_score_gemma":0.002378793,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003648621,"about_ca_topic_score_gemma":0.007052116,"domain_scores_codex":[0.997617,0.001222392,0.0002441988,0.0005326098,0.0002710345,0.0001128239],"domain_scores_gemma":[0.9925287,0.005347686,0.0002834562,0.0008090024,0.0007260548,0.0003051313],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002468824,0.0008603784,0.009751574,0.003582824,0.0002510039,0.0009073474,0.002419348,0.1476026,0.0270733,0.02397025,0.05875326,0.7223592],"study_design_scores_gemma":[0.0003001214,0.0003360963,0.0007177265,0.0001028192,0.00007971963,0.0002129471,0.0004091675,0.9192589,0.01101906,0.0424304,0.02507843,0.00005457685],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03276921,0.001357534,0.8682693,0.001650049,0.0002321498,0.000708088,0.003238614,0.0896019,0.002173261],"genre_scores_gemma":[0.3061348,0.0004624226,0.6806583,0.0009766818,0.0001140405,0.0008365124,0.007307806,0.001140973,0.00236852],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.00774077,"threshold_uncertainty_score":0.02728021,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01004011921151553,"score_gpt":0.2772171588428123,"score_spread":0.2671770396312968,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}