{"id":"W2041542884","doi":"10.1007/s10459-007-9068-0","title":"Undesired variance due to examiner stringency/leniency effect in communication skill scores assessed in OSCEs","year":2007,"lang":"en","type":"article","venue":"Advances in Health Sciences Education","topic":"Innovations in Medical Education","field":"Medicine","cited_by":109,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Variance (accounting); Rasch model; Objective structured clinical examination; Raw score; Psychology; Medical education; Communication skills; Applied psychology; Medicine; Statistics; Developmental psychology; Mathematics; Raw data","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02923321,0.0007487002,0.000954552,0.001243447,0.000964828,0.001089939,0.0008561356,0.001047043,0.003973038],"category_scores_gemma":[0.0957023,0.0005210573,0.001337927,0.001350438,0.001196613,0.0007993254,0.001716082,0.001409217,0.0007614698],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005689814,"about_ca_system_score_gemma":0.0006636063,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00212231,"about_ca_topic_score_gemma":0.003555723,"domain_scores_codex":[0.9661147,0.01218391,0.003534018,0.008203264,0.008605315,0.001358832],"domain_scores_gemma":[0.8659548,0.1020963,0.005234569,0.01891192,0.006489645,0.00131278],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.003421199,0.000788783,0.8813214,0.0004318765,0.002248428,0.0009025107,0.003617651,0.002436948,0.02475371,0.0026761,0.002168987,0.07523243],"study_design_scores_gemma":[0.00005955781,0.0005301621,0.9849597,0.00002800233,0.0004425085,0.0004231431,0.0001748558,0.003810829,0.007517244,0.0008869295,0.001137937,0.00002927924],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9425771,0.0008441292,0.04922818,0.0002578982,0.0005914484,0.0003841958,0.0008009,0.0003625347,0.00495367],"genre_scores_gemma":[0.9918458,0.0000686023,0.005349555,0.0001528518,0.00009840763,0.0001804447,0.0004121473,0.0002283995,0.001663862],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9707668,"threshold_uncertainty_score":0.1546018,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0203232632687586,"score_gpt":0.4305531748638134,"score_spread":0.4102299115950548,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}