{"id":"W3017068696","doi":"10.26822/iejee.2020459464","title":"Investigation of Rater Tendencies and Reliability in Different Assessment Methods with Many Facet Rasch Model","year":2020,"lang":"en","type":"article","venue":"lnternational Electronic Journal of Elementary Education","topic":"Education and Critical Thinking Development","field":"Social Sciences","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Office of International Science and Engineering; University of Toronto","keywords":"Rasch model; Facet (psychology); Psychology; Inter-rater reliability; Reliability (semiconductor); Polytomous Rasch model; Test validity; Psychometrics; Rating scale; Item response theory; Clinical psychology; Developmental psychology; Social psychology; Big Five personality traits","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1601514,0.0009896595,0.001113521,0.00325373,0.001081808,0.001612514,0.001011442,0.0009038924,0.0007641831],"category_scores_gemma":[0.2781256,0.0005998417,0.001895809,0.003044766,0.001607074,0.001420203,0.001660765,0.001036887,0.0006099426],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008124881,"about_ca_system_score_gemma":0.001030066,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001064587,"about_ca_topic_score_gemma":0.001187665,"domain_scores_codex":[0.7991601,0.1335689,0.01701209,0.01238871,0.03639081,0.001479479],"domain_scores_gemma":[0.5832229,0.2941029,0.02169298,0.0302989,0.06975535,0.000927036],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00295221,0.0007223163,0.6249965,0.001598509,0.002880831,0.0004136941,0.04687383,0.01033106,0.01583839,0.005547851,0.002224452,0.2856204],"study_design_scores_gemma":[0.0003594562,0.007523059,0.7657176,0.001135895,0.002120616,0.002155108,0.01604277,0.1373059,0.03704171,0.009942526,0.02008401,0.0005714723],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6942953,0.001635887,0.2909108,0.0002705995,0.0003765959,0.002929736,0.0003981916,0.0005915574,0.008591425],"genre_scores_gemma":[0.8976223,0.000275365,0.09761675,0.00007079975,0.0000732124,0.002499923,0.0003739252,0.0001617868,0.001305911],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1601514,"threshold_uncertainty_score":0.8469716,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03179235665838372,"score_gpt":0.3828931222103248,"score_spread":0.3511007655519411,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}