{"id":"W3127624731","doi":"10.1038/s41405-021-00067-4","title":"The interrelationship between confidence and correctness in a multiple-choice assessment: pointing out misconceptions and assuring valuable questions","year":2021,"lang":"en","type":"article","venue":"BDJ Open","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Correctness; Low Confidence; Confidence interval; Psychology; Point (geometry); Mathematics education; Sample (material); Computer science; Social psychology; Statistics; Mathematics; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01107653,0.0003858103,0.0004059132,0.001227651,0.0003678983,0.001577008,0.0006201506,0.0007775541,0.002474885],"category_scores_gemma":[0.08010285,0.0003472554,0.000547328,0.0005874909,0.0009600254,0.001280118,0.001117941,0.001006577,0.0003619917],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004622121,"about_ca_system_score_gemma":0.0006827475,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007110914,"about_ca_topic_score_gemma":0.0008532969,"domain_scores_codex":[0.9900718,0.004346723,0.001341353,0.0006144235,0.003197333,0.0004283291],"domain_scores_gemma":[0.8582979,0.09174111,0.03547632,0.002312301,0.009218788,0.002953583],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0002602704,0.0003115587,0.9823905,0.0001211997,0.00006028844,0.00006201941,0.003090407,0.000142352,0.0010226,0.00006370341,0.0001051477,0.01236998],"study_design_scores_gemma":[0.0000224774,0.001074267,0.9897975,0.0001028072,0.00004576826,0.0005544277,0.003952044,0.0017264,0.0019299,0.0002483073,0.0005047272,0.00004142467],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.998444,0.00009515531,0.000838757,0.00006607692,0.000006404459,0.00002732451,0.00002734343,0.000007849445,0.0004870309],"genre_scores_gemma":[0.9988006,0.00005317277,0.0009647094,0.00002388158,0.000005509504,0.00002706063,0.0000284565,0.000002228437,0.00009437001],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01107653,"threshold_uncertainty_score":0.05857897,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1385446853540866,"score_gpt":0.4454390641274341,"score_spread":0.3068943787733475,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}