{"id":"W2478762182","doi":"10.4300/jgme-d-15-00751.1","title":"Assessing the Reliability of Performance Assessment Scores: Some Considerations in Selecting an Appropriate Framework","year":2016,"lang":"en","type":"article","venue":"Journal of Graduate Medical Education","topic":"Innovations in Medical Education","field":"Medicine","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Medical Council of Canada","funders":"","keywords":"Accreditation; Observational study; Reliability (semiconductor); Graduate medical education; Sample (material); Psychology; Test (biology); Process (computing); Computer science; Educational measurement; Domain (mathematical analysis); Sample size determination; Medical education; Applied psychology; Medicine; Statistics; Curriculum; Mathematics; Pedagogy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.5823919,0.002709356,0.007354531,0.0156649,0.004621178,0.01133597,0.01119297,0.008079582,0.001152622],"category_scores_gemma":[0.7241778,0.001940923,0.004949512,0.01069263,0.01661718,0.008602833,0.01033072,0.01297062,0.0007826512],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.008252094,"about_ca_system_score_gemma":0.01425825,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01229831,"about_ca_topic_score_gemma":0.01622865,"domain_scores_codex":[0.4429073,0.4084235,0.06215107,0.009844407,0.07382853,0.002845178],"domain_scores_gemma":[0.1755581,0.6756867,0.02342918,0.02301,0.09988818,0.002427891],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.001720357,0.0009853367,0.2545773,0.007208771,0.002451793,0.001956355,0.02308813,0.009366168,0.001780583,0.1289513,0.03859567,0.5293183],"study_design_scores_gemma":[0.001055778,0.006352973,0.3144901,0.06546971,0.002543079,0.005573856,0.01819277,0.1156029,0.005263904,0.2775909,0.1861512,0.001712843],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.07043304,0.07932378,0.702404,0.1007754,0.004512393,0.01157562,0.001379167,0.0007757719,0.02882096],"genre_scores_gemma":[0.3486923,0.007030669,0.6167708,0.009198985,0.003633816,0.01246321,0.0008495075,0.0004247128,0.0009359334],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.4176081,"threshold_uncertainty_score":0.5149852,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04909405547366617,"score_gpt":0.4256244603539484,"score_spread":0.3765304048802822,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}