{"id":"W3017068696","doi":"10.26822/iejee.2020459464","title":"Investigation of Rater Tendencies and Reliability in Different Assessment Methods with Many Facet Rasch Model","year":2020,"lang":"en","type":"article","venue":"lnternational Electronic Journal of Elementary Education","topic":"Education and Critical Thinking Development","field":"Social Sciences","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Office of International Science and Engineering; University of Toronto","keywords":"Rasch model; Facet (psychology); Psychology; Inter-rater reliability; Reliability (semiconductor); Polytomous Rasch model; Test validity; Psychometrics; Rating scale; Item response theory; Clinical psychology; Developmental psychology; Social psychology; Big Five personality traits","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00149616,0.00008834949,0.0001670531,0.0001129205,0.00007397165,0.00003190618,0.0001540858,0.00003173055,0.00009258406],"category_scores_gemma":[0.0001265957,0.00007076483,0.00003106701,0.0001598439,0.0001063495,0.0003189547,0.00001840525,0.0002486178,2.859828e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006338579,"about_ca_system_score_gemma":0.002780707,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001453458,"about_ca_topic_score_gemma":0.0001129131,"domain_scores_codex":[0.9983813,0.0002954023,0.0004590317,0.0001359241,0.0005244937,0.0002038777],"domain_scores_gemma":[0.9991414,0.0001008642,0.0002617221,0.00004320892,0.0003350779,0.0001177389],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001599875,0.0005722875,0.5001336,0.000107533,0.0001553858,7.078418e-7,0.04046801,0.001410376,0.006311338,0.438084,0.0007367278,0.01186002],"study_design_scores_gemma":[0.001760702,0.0007584043,0.5627235,0.0002513712,0.0001153995,0.00001076876,0.03175276,0.00743155,0.004953508,0.3871566,0.002702028,0.000383419],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9353642,0.0001409088,0.02312027,0.04072104,0.0001078056,0.0001794933,0.000001541509,0.000004510756,0.0003601717],"genre_scores_gemma":[0.9645005,0.0002609141,0.03386724,0.001144369,0.0001114382,0.0000129019,0.00001181272,0.000005221703,0.00008558844],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.06258994,"threshold_uncertainty_score":0.4932855,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03179235665838372,"score_gpt":0.3828931222103248,"score_spread":0.3511007655519411,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}