{"id":"W2148427497","doi":"10.7202/1025007ar","title":"The versatility of generalizability theory as a tool for exploring and controlling measurement error","year":2014,"lang":"en","type":"article","venue":"Mesure et évaluation en éducation","topic":"Educational Assessment and Improvement","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Generalizability theory; Computer science; Sample (material); Observational error; Estimation; Item response theory; Econometrics; Statistics; Psychology; Mathematics; Psychometrics; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.1069501,0.0001031701,0.0001697009,0.00007213799,0.0002288855,0.0001385404,0.0002108161,0.00003520639,0.00005366791],"category_scores_gemma":[0.02568284,0.00006776433,0.00007514031,0.0001576038,0.00006001338,0.000378463,0.00003056768,0.00005785191,0.000006403904],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001629032,"about_ca_system_score_gemma":0.0004784453,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002834154,"about_ca_topic_score_gemma":0.00006498655,"domain_scores_codex":[0.9941108,0.003261387,0.0006841619,0.0003358983,0.001470022,0.0001376791],"domain_scores_gemma":[0.9909674,0.006657439,0.0003903017,0.0004280436,0.001519553,0.00003729519],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0005749511,0.0005457999,0.0827184,0.00008292776,0.0001670588,1.260925e-8,0.09234915,0.001780538,0.02234375,0.3040437,0.003025252,0.4923684],"study_design_scores_gemma":[0.0007775382,0.0001490547,0.6517983,0.00001447155,0.00005065889,2.11633e-7,0.003966501,0.01343062,0.003398944,0.3184723,0.007823982,0.0001174448],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8781407,0.00009844188,0.1109006,0.008820708,0.0008581773,0.0008322247,0.000004196247,0.00001198186,0.000332998],"genre_scores_gemma":[0.9970794,0.00001456122,0.001984498,0.0001974136,0.0001092083,0.0003914397,0.00001170452,0.000006504991,0.0002052951],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5690799,"threshold_uncertainty_score":0.9825242,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3700201145714859,"score_gpt":0.4653012208777185,"score_spread":0.0952811063062326,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}