{"id":"W4313410276","doi":"10.1111/emip.12537","title":"Using Active Learning Methods to Strategically Select Essays for Automated Scoring","year":2022,"lang":"en","type":"article","venue":"Educational Measurement Issues and Practice","topic":"Innovative Teaching and Learning Methods","field":"Psychology","cited_by":19,"is_retracted":false,"has_abstract":true,"ca_institutions":"Centre for Advancing Health Outcomes; University of Alberta","funders":"","keywords":"Computer science; Artificial intelligence; Machine learning; Scalability; Active learning (machine learning); Encoder; Transformer; Database; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009012679,0.001024253,0.0008275858,0.002218383,0.0005658511,0.002347926,0.002028486,0.0009438706,0.002421166],"category_scores_gemma":[0.03578798,0.0003481179,0.0004752836,0.001017007,0.0007611456,0.002339405,0.001725329,0.001427078,0.00128935],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008422752,"about_ca_system_score_gemma":0.001141909,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001038913,"about_ca_topic_score_gemma":0.002240082,"domain_scores_codex":[0.9945723,0.002951531,0.0003483021,0.0006740714,0.001276304,0.0001774001],"domain_scores_gemma":[0.9594585,0.02750979,0.00269479,0.002420557,0.007185963,0.0007304285],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005755164,0.0006788155,0.009924227,0.0001435319,0.00008282861,0.00007601165,0.000559862,0.07586161,0.01485853,0.005712437,0.002231063,0.8892955],"study_design_scores_gemma":[0.00007734585,0.000261923,0.002011831,0.00002379124,0.00002185136,0.00005037778,0.0001745035,0.9686193,0.0202601,0.006826773,0.001642187,0.00002993722],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1346761,0.000138181,0.8590714,0.0002653608,0.00008059505,0.0003486085,0.00011421,0.002285823,0.003019686],"genre_scores_gemma":[0.6561729,0.00005552311,0.3396644,0.00008356924,0.00004757701,0.000362164,0.0002587212,0.0001744585,0.003180752],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009012679,"threshold_uncertainty_score":0.04766423,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3365902019122547,"score_gpt":0.5652165547572977,"score_spread":0.228626352845043,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}