{"id":"W4404851119","doi":"10.1016/j.eswa.2024.125924","title":"Pairwise dual-level alignment for cross-prompt automated essay scoring","year":2024,"lang":"en","type":"article","venue":"Expert Systems with Applications","topic":"Topic Modeling","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"ca_institutions":"Canada Research Chairs; University of Toronto","funders":"Taishan Scholar Project of Shandong Province; Natural Science Foundation of Shandong Province; National Natural Science Foundation of China","keywords":"Pairwise comparison; Dual (grammatical number); Computer science; Artificial intelligence; Cross-validation; Data mining; Natural language processing; Pattern recognition (psychology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004572551,0.00156632,0.001619032,0.003968651,0.002037146,0.003288377,0.00244906,0.002174348,0.0171283],"category_scores_gemma":[0.0222309,0.000935581,0.001002503,0.003883335,0.0005705011,0.003068874,0.005085815,0.003268929,0.01680915],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00073869,"about_ca_system_score_gemma":0.002787334,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001734921,"about_ca_topic_score_gemma":0.004564425,"domain_scores_codex":[0.9922038,0.003142156,0.000636026,0.002072296,0.001310861,0.0006348235],"domain_scores_gemma":[0.9867264,0.004715504,0.0007212324,0.002344179,0.004799773,0.0006928652],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000962982,0.000384468,0.003875869,0.0005566932,0.0001575169,0.0002393947,0.0009241128,0.006295388,0.05327974,0.007709171,0.04031002,0.8853047],"study_design_scores_gemma":[0.0002841105,0.0007982772,0.01098355,0.0002478831,0.0002960244,0.001025183,0.001976294,0.7378871,0.1086678,0.05312553,0.08444628,0.0002618483],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02439539,0.0005568009,0.9445657,0.0002094331,0.0004097023,0.0002898555,0.002119701,0.02171845,0.005734981],"genre_scores_gemma":[0.2653123,0.0002071815,0.7112133,0.0001748888,0.0002247024,0.0006217745,0.01094667,0.003006677,0.008292479],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.0171283,"threshold_uncertainty_score":0.05729985,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04273792357509929,"score_gpt":0.3207816349410548,"score_spread":0.2780437113659555,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}