{"id":"W2164610426","doi":"10.1177/0163278704267043","title":"Standardized Assessment of Reasoning in Contexts of Uncertainty","year":2004,"lang":"en","type":"article","venue":"Evaluation & the Health Professions","topic":"Clinical Reasoning and Diagnostic Skills","field":"Medicine","cited_by":153,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Psychology; Standardized test; Medical physics; Computer science; Natural language processing; Medical education; Mathematics education; Medicine","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008924841,0.0004959799,0.0007922322,0.003345649,0.0004517167,0.001466891,0.0006789284,0.0006150065,0.001860144],"category_scores_gemma":[0.03761058,0.0002155723,0.0007728306,0.001576964,0.001376806,0.001794133,0.002114507,0.0009496404,0.0004140743],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008435542,"about_ca_system_score_gemma":0.002585997,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000694291,"about_ca_topic_score_gemma":0.001468102,"domain_scores_codex":[0.9885834,0.006035696,0.002157586,0.0005238253,0.002351228,0.0003482412],"domain_scores_gemma":[0.9782613,0.009486073,0.004155952,0.001885868,0.005358221,0.0008526546],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001940137,0.001629068,0.2493794,0.002069063,0.0003002216,0.001042026,0.01122279,0.008847052,0.02251993,0.02244157,0.007561297,0.6710474],"study_design_scores_gemma":[0.0007091098,0.01184583,0.720677,0.001913255,0.0004310931,0.006771906,0.01875359,0.046841,0.05962495,0.08833673,0.04345666,0.0006388666],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8314853,0.001909932,0.1325131,0.0006171973,0.0002874424,0.005166412,0.0009864346,0.0005617639,0.02647251],"genre_scores_gemma":[0.8751742,0.001336359,0.1174494,0.0001244338,0.00006323108,0.00255569,0.001031777,0.0000390499,0.002225812],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008924841,"threshold_uncertainty_score":0.04719967,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1004936907630079,"score_gpt":0.5254536042443122,"score_spread":0.4249599134813042,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}