{"id":"W107454019","doi":"10.55016/ojs/ajer.v49i1.54961","title":"An Investigation of the Accuracy of Alternative Methods of True Score Estimation in High-Stakes Mixed-Format Examinations","year":2003,"lang":"en","type":"article","venue":"Alberta Journal of Educational Research","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Statistics; Estimation; Psychology; Mathematics education; Econometrics; Mathematics; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1562006,0.001086611,0.0009790116,0.003735138,0.0009554552,0.003536168,0.003131992,0.002242972,0.001103635],"category_scores_gemma":[0.6381663,0.0008562616,0.001552733,0.002811409,0.003115093,0.003174746,0.003180416,0.002046135,0.000737663],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002211133,"about_ca_system_score_gemma":0.001597013,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006398714,"about_ca_topic_score_gemma":0.007322633,"domain_scores_codex":[0.782743,0.1742211,0.008807516,0.009278876,0.02376994,0.001179588],"domain_scores_gemma":[0.2095933,0.707186,0.02181529,0.03765758,0.02295963,0.0007882111],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.004055869,0.0005067388,0.615566,0.0005614281,0.001812966,0.0002211126,0.005362286,0.02254298,0.002891609,0.008858275,0.001213504,0.3364072],"study_design_scores_gemma":[0.0006447081,0.004153352,0.5109055,0.0006968862,0.001203101,0.002603997,0.002916211,0.4306453,0.02328505,0.01740135,0.005019352,0.0005252022],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.797231,0.001962638,0.1928936,0.001250282,0.0001899599,0.0004709138,0.0004246838,0.0004269146,0.005149944],"genre_scores_gemma":[0.900359,0.0004094457,0.09769739,0.0001891685,0.0000576494,0.0001799115,0.0003003573,0.00008061933,0.0007263896],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1562006,"threshold_uncertainty_score":0.8260777,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6078835678103994,"score_gpt":0.5942529116662607,"score_spread":0.01363065614413872,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}