{"id":"W3172809335","doi":"10.5430/ijhe.v10n6p22","title":"The Effect of Multiple-Choice Test Items’ Difficulty Degree on the Reliability Coefficient and the Standard Error of Measurement Depending on the Item Response Theory (IRT)","year":2021,"lang":"en","type":"article","venue":"International Journal of Higher Education","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Item response theory; Statistics; Reliability (semiconductor); Degree (music); Test (biology); Standard error; Mathematics; Function (biology); Psychometrics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.0848599,0.001353418,0.001382044,0.002215923,0.0007044107,0.001782203,0.001180819,0.001252271,0.001228294],"category_scores_gemma":[0.3210537,0.000798814,0.002467107,0.002050454,0.002107365,0.00212429,0.001973676,0.001899558,0.0005372029],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008777719,"about_ca_system_score_gemma":0.0008375411,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001026862,"about_ca_topic_score_gemma":0.001081002,"domain_scores_codex":[0.8437092,0.08746118,0.01202593,0.01687439,0.03822871,0.001700549],"domain_scores_gemma":[0.3481972,0.578693,0.02350339,0.0264156,0.02221218,0.0009785445],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001964508,0.0004445679,0.8533931,0.0008128763,0.003714187,0.0003471728,0.006238482,0.006379262,0.008837263,0.002771856,0.001267477,0.1138293],"study_design_scores_gemma":[0.0001590198,0.003298645,0.9392226,0.0005092628,0.00166371,0.001698491,0.001917807,0.02667209,0.01648133,0.004740441,0.003329558,0.0003068765],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8384287,0.003480544,0.1495756,0.0004134441,0.0004145927,0.0004178811,0.0003618089,0.0004015758,0.006505857],"genre_scores_gemma":[0.9803345,0.0002419659,0.01806489,0.0001067208,0.00006983813,0.0002559659,0.0002529013,0.0001324511,0.0005407684],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9151401,"threshold_uncertainty_score":0.4487874,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2437536790605494,"score_gpt":0.4476650472345339,"score_spread":0.2039113681739845,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}