{"id":"W2019936193","doi":"10.1177/0163278713475868","title":"Evidence for the Validity of Grouped Self-Assessments in Measuring the Outcomes of Educational Programs","year":2013,"lang":"en","type":"article","venue":"Evaluation & the Health Professions","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":26,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan","funders":"University of Saskatchewan","keywords":"Psychology; Correlation; Self-assessment; Applied psychology; Clinical psychology; Social psychology; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2809293,0.0009126329,0.001195141,0.006641868,0.001346771,0.004152265,0.003133726,0.001668018,0.002065654],"category_scores_gemma":[0.5188504,0.0007548239,0.002669506,0.00890469,0.006230731,0.004747943,0.003380226,0.001875641,0.0008809696],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002209197,"about_ca_system_score_gemma":0.002934021,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006060313,"about_ca_topic_score_gemma":0.008781861,"domain_scores_codex":[0.7354783,0.1894371,0.01348967,0.01110373,0.04893159,0.001559578],"domain_scores_gemma":[0.1871417,0.6572934,0.04897281,0.05137894,0.05273573,0.002477536],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.002398466,0.001103986,0.7082267,0.005853082,0.007585355,0.0001021683,0.007957241,0.003758685,0.000535763,0.009622407,0.002215206,0.2506409],"study_design_scores_gemma":[0.0002872772,0.004387881,0.9415088,0.007106867,0.002702236,0.000392503,0.004183336,0.00965141,0.003043536,0.01453409,0.01197997,0.0002220412],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.797658,0.03165706,0.09593078,0.004558052,0.001221517,0.001630795,0.002027848,0.0002478438,0.06506808],"genre_scores_gemma":[0.9733396,0.00310725,0.02096557,0.0005290544,0.000254064,0.0004338171,0.0005498764,0.00008827569,0.0007324924],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2809293,"threshold_uncertainty_score":0.8867422,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.7640554176157064,"score_gpt":0.6509811233514293,"score_spread":0.1130742942642771,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}