{"id":"W7077933530","doi":"10.48448/ce7k-8233","title":"Rationale Behind Essay Scores: Enhancing S-LLM's Multi-Trait Essay Scoring with Rationale Generated by LLMs","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Rubric; Benchmark (surveying); Trait; Language model; Code (set theory); Scoring system; Qualitative property","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0009500346,0.0004761157,0.0003986475,0.0004543056,0.0006604629,0.0005990518,0.00219425,0.000252807,0.0006618418],"category_scores_gemma":[0.0004201186,0.0004110585,0.00005892012,0.001259807,0.0005406697,0.0005548956,0.0005237547,0.0004358573,0.0000697672],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002256994,"about_ca_system_score_gemma":0.003946108,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002586069,"about_ca_topic_score_gemma":0.0007361771,"domain_scores_codex":[0.9962649,0.00007135138,0.0004609875,0.001484512,0.001017753,0.0007005192],"domain_scores_gemma":[0.9977964,0.0001776801,0.0003986814,0.000899214,0.0005200162,0.0002080418],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006292514,0.002647476,0.002593143,0.001578887,0.0006240272,0.000314057,0.005682262,0.04259516,0.3738475,0.1716685,0.3643724,0.03401362],"study_design_scores_gemma":[0.002905993,0.0003193986,0.0001791352,0.002678062,0.00009461253,0.0001076485,0.0004425079,0.4482787,0.1884882,0.004258191,0.348936,0.00331156],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.0001355781,0.0004740038,0.8032193,0.001895863,0.0006664684,0.0007054098,0.00007752644,0.0004533969,0.1923725],"genre_scores_gemma":[0.01751257,0.00003353078,0.1815644,0.000783401,0.000384623,0.0001163128,0.0003434792,0.00005217865,0.7992095],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.6216549,"threshold_uncertainty_score":0.9998341,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01635607734806291,"score_gpt":0.2471395748755525,"score_spread":0.2307834975274896,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}