{"id":"W4392971412","doi":"10.1111/medu.15374","title":"‘Good’ evaluation: Methodological diversity may be empress, but sound methods remain queen","year":2024,"lang":"en","type":"letter","venue":"Medical Education","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"The Wilson Centre; University of Toronto; Centre for Addiction and Mental Health","funders":"","keywords":"Queen (butterfly); Diversity (politics); Sound (geography); Sociology; Anthropology; Biology; Ecology; Acoustics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1204777,0.0008096521,0.001498619,0.001548953,0.0070798,0.01450845,0.003557565,0.03631776,0.01077997],"category_scores_gemma":[0.2677021,0.0008018406,0.001797725,0.001582985,0.02047495,0.01501581,0.005449737,0.05268258,0.005035934],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01258645,"about_ca_system_score_gemma":0.0256214,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01830092,"about_ca_topic_score_gemma":0.0347243,"domain_scores_codex":[0.8663853,0.08479834,0.01009906,0.007058038,0.02849993,0.003159355],"domain_scores_gemma":[0.6159699,0.2952605,0.008599821,0.009348705,0.05673568,0.01408546],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00004978384,0.00002561448,0.0006585455,0.0001639169,0.00002164493,0.0001672908,0.0007894302,0.00006687792,0.000141279,0.02756199,0.9524618,0.01789191],"study_design_scores_gemma":[0.0001238135,0.00008236654,0.001916506,0.001737245,0.00003428945,0.0006617337,0.002901175,0.0008598517,0.0003059629,0.09645041,0.8947809,0.0001457811],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.00009350078,0.0005716872,0.0003001436,0.9953184,0.002227414,0.000008511178,0.0000104393,0.00001146499,0.001458373],"genre_scores_gemma":[0.00346239,0.0005691979,0.001243344,0.9857932,0.005151717,0.00005070258,0.00001235923,0.00003811354,0.003678993],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.8795223,"threshold_uncertainty_score":0.6371544,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5639421722179152,"score_gpt":0.6388402463653121,"score_spread":0.07489807414739691,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}