{"id":"W4416578240","doi":"10.2196/82487","title":"Assessment of Physician Preferences for Large Language Model–Generated Responses Across Geographic Regions and Clinical Experience Levels: Preliminary Survey Study","year":2025,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Consistency (knowledge bases); Exploratory research; Workflow; Survey research; MEDLINE; Survey data collection; Patient experience; Preference","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01569811,0.0001937471,0.0002751031,0.001146952,0.0004170211,0.0009166137,0.000338314,0.0005913028,0.002607272],"category_scores_gemma":[0.04807613,0.0002158214,0.0005112737,0.0009173772,0.000576999,0.001010462,0.0009282058,0.0005829406,0.0006311742],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007656311,"about_ca_system_score_gemma":0.0007426101,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001312101,"about_ca_topic_score_gemma":0.00178148,"domain_scores_codex":[0.9896833,0.006566905,0.001208169,0.0007596566,0.001340739,0.0004411659],"domain_scores_gemma":[0.9233527,0.0500546,0.01359409,0.002310972,0.007830131,0.00285755],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0004041537,0.0003971452,0.9650097,0.0001685075,0.00009007668,0.0001382166,0.01345454,0.0002436855,0.0008453311,0.0001002728,0.00114325,0.01800517],"study_design_scores_gemma":[0.00009857569,0.002520656,0.9684486,0.0001138986,0.00005212075,0.000588731,0.02198252,0.002377571,0.0006223809,0.0001547043,0.002995675,0.00004462718],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9982966,0.00006521637,0.0004091811,0.0001882296,0.000004253331,0.00008747503,0.0002551026,0.00001016294,0.0006837671],"genre_scores_gemma":[0.9981782,0.00008660634,0.0009783633,0.0002315775,0.000008893855,0.0001578366,0.0001816531,0.000006386502,0.0001704181],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01569811,"threshold_uncertainty_score":0.08302057,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5285633264259062,"score_gpt":0.6589013764727935,"score_spread":0.1303380500468873,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}