{"id":"W2923725345","doi":"10.3389/feduc.2019.00020","title":"Development and Examination of a Tool to Assess Score Report Quality","year":2019,"lang":"en","type":"article","venue":"Frontiers in Education","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Generalizability theory; Reliability (semiconductor); Scale (ratio); Quality (philosophy); Stakeholder; Rating scale; Standards for Educational and Psychological Testing; Psychology; Sample (material); Applied psychology; Accountability; Medical education; Medicine; Political science; Higher education; Public relations; Geography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006772508,0.00006004625,0.0001641202,0.0003593472,0.0000280806,0.00004991001,0.0001406082,0.00003793166,0.0001471538],"category_scores_gemma":[0.0009654555,0.00005194149,0.0000145093,0.000494643,0.00001426553,0.0003272061,0.00003422481,0.00004573436,0.00003109921],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001491182,"about_ca_system_score_gemma":0.0007158009,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001454128,"about_ca_topic_score_gemma":0.00002651973,"domain_scores_codex":[0.9979758,0.0001426314,0.0007066407,0.0002603375,0.0008300351,0.00008457692],"domain_scores_gemma":[0.9989453,0.00009532728,0.0002947349,0.0002977524,0.0003281984,0.00003871413],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000007578385,0.00005450327,0.6113175,0.00001228195,0.000002253156,9.930377e-8,0.003798033,0.00003952871,0.0002044974,0.0006347283,0.002361386,0.3815676],"study_design_scores_gemma":[0.0001212844,0.00001659402,0.9610681,0.0000293987,0.0000014723,0.000001568931,0.01008394,0.0004135921,0.001025515,0.0009446161,0.02621934,0.00007458645],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9837173,0.00003777545,0.007769759,0.0003117715,0.001705707,0.0003480961,6.976279e-7,0.000005090054,0.006103844],"genre_scores_gemma":[0.9261335,0.000004360299,0.07045294,0.000153732,0.00002091091,0.0000408453,0.00001430398,0.00000301713,0.00317639],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.381493,"threshold_uncertainty_score":0.234723,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2202473128127325,"score_gpt":0.5037553939418976,"score_spread":0.2835080811291651,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}