{"id":"W4315850919","doi":"10.53841/bpsadm.2017.9.1.25","title":"Can combining assessments improve test fairness and enhance user experience?","year":2017,"lang":"en","type":"article","venue":"Assessment and Development Matters","topic":"Human-Automation Interaction and Safety","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Canadian Association of Physicists","funders":"","keywords":"Test (biology); Computer science; Reliability engineering; Psychology; Process management; Business; Engineering; Geology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0002147615,0.0002053115,0.0002099952,0.00008244734,0.0008227753,0.0004623756,0.0002062412,0.0000681346,0.001178457],"category_scores_gemma":[0.00001231852,0.0001948905,0.00001988282,0.00002506284,0.0001294304,0.0003906044,0.0001635626,0.0001574205,0.00006961411],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008389878,"about_ca_system_score_gemma":0.00007281473,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008344844,"about_ca_topic_score_gemma":0.0000524213,"domain_scores_codex":[0.9986938,0.00004117193,0.0003394464,0.0004254501,0.000209267,0.0002908213],"domain_scores_gemma":[0.9990878,0.0001043364,0.0002751978,0.0003494506,0.0000497583,0.0001334434],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00002850351,0.0002141973,0.9296501,0.00004442972,0.000148859,0.00002951347,0.01332609,2.48529e-7,0.002047525,0.003535425,0.008348208,0.04262686],"study_design_scores_gemma":[0.0009013723,0.00005016346,0.9676856,0.00004847093,0.000008499474,0.00001111188,0.004172585,0.00004899002,0.0006406743,0.00006831251,0.02606769,0.0002964669],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9718837,0.00001214216,0.002430372,0.003657497,0.001567227,0.0002847494,0.000006515882,0.00008975186,0.02006801],"genre_scores_gemma":[0.9876377,0.00001053638,0.002793699,0.001885788,0.00003193344,0.0001703967,0.0000174245,0.00001865841,0.007433866],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04233039,"threshold_uncertainty_score":0.9997346,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03281881503818991,"score_gpt":0.4140897758775974,"score_spread":0.3812709608394075,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}