{"id":"W4404851119","doi":"10.1016/j.eswa.2024.125924","title":"Pairwise dual-level alignment for cross-prompt automated essay scoring","year":2024,"lang":"en","type":"article","venue":"Expert Systems with Applications","topic":"Topic Modeling","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":false,"ca_institutions":"Canada Research Chairs; University of Toronto","funders":"Taishan Scholar Project of Shandong Province; Natural Science Foundation of Shandong Province; National Natural Science Foundation of China","keywords":"Pairwise comparison; Dual (grammatical number); Computer science; Artificial intelligence; Cross-validation; Data mining; Natural language processing; Pattern recognition (psychology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000316746,0.0001788114,0.000178715,0.0001006221,0.0002760178,0.0006717787,0.0005029907,0.00006569753,0.000002922738],"category_scores_gemma":[0.000007529146,0.0001445643,0.00005509293,0.0003511344,0.00003295899,0.0003513012,0.0001078251,0.00007263134,0.00007949213],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001952586,"about_ca_system_score_gemma":0.0001884272,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000117215,"about_ca_topic_score_gemma":0.000004384996,"domain_scores_codex":[0.9983238,0.00002573298,0.0003557758,0.0006831416,0.0002911874,0.0003203641],"domain_scores_gemma":[0.9986655,0.00011828,0.00006818057,0.0009161557,0.0001128395,0.000119046],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001042864,0.0001884276,0.0003143614,0.0008300255,0.0002220132,0.00002429532,0.007958396,0.02196006,0.008402274,0.9222432,0.01368937,0.02415717],"study_design_scores_gemma":[0.0002203462,0.00002927705,0.00006456596,0.0001973983,0.000005684128,0.00004691229,0.00009286198,0.8933487,0.001014358,0.0003800152,0.1043669,0.0002329764],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0008474001,0.001765309,0.9916075,0.0006920079,0.0004892162,0.001696426,0.00002103261,0.002257967,0.0006231246],"genre_scores_gemma":[0.8611422,0.000008411314,0.1241403,0.00009330131,0.0004781157,0.01164912,0.00001726382,0.00003975231,0.002431574],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9218631,"threshold_uncertainty_score":0.6477978,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04273792357509929,"score_gpt":0.3207816349410548,"score_spread":0.2780437113659555,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}