{"id":"W4415964192","doi":"10.48550/arxiv.2510.17389","title":"EduAdapt: A Question Answer Benchmark Dataset for Evaluating Grade-Level Adaptability in LLMs","year":2025,"lang":"","type":"preprint","venue":"ArXiv.org","topic":"Intelligent Tutoring Systems and Adaptive Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; Social Sciences and Humanities Research Council of Canada; Natural Sciences and Engineering Research Council of Canada; Canada Research Chairs","keywords":"Adaptability; Benchmark (surveying); Set (abstract data type); Vocabulary; Code (set theory); Range (aeronautics)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.01061648,0.001215963,0.001508851,0.0006139225,0.0007817564,0.000554713,0.002885918,0.0008557766,0.0001123488],"category_scores_gemma":[0.003025479,0.001314749,0.00056504,0.001005773,0.0001941915,0.0009709563,0.002972058,0.002406602,0.0001289527],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001111611,"about_ca_system_score_gemma":0.001708257,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004600996,"about_ca_topic_score_gemma":0.0006665548,"domain_scores_codex":[0.9890081,0.001899746,0.002647737,0.003694821,0.001179386,0.001570198],"domain_scores_gemma":[0.9927204,0.001709228,0.001324947,0.002967321,0.0009814815,0.0002966838],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004811931,0.001942201,0.7119358,0.00507174,0.0006283199,0.00008141405,0.00983354,0.102986,0.002223159,0.08035641,0.00318169,0.08127858],"study_design_scores_gemma":[0.001966792,0.0009141034,0.4088088,0.009440237,0.0002457933,0.00002182391,0.0009366444,0.4566425,0.001691107,0.003980169,0.1125088,0.002843261],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3533072,0.001321167,0.6251502,0.001253545,0.01089319,0.00458236,0.002981643,0.000158622,0.0003520875],"genre_scores_gemma":[0.947815,0.0001120304,0.03896403,0.000291939,0.001177719,0.001044703,0.002300401,0.00006632238,0.008227848],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.5945078,"threshold_uncertainty_score":0.9998949,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2441734194768091,"score_gpt":0.4039861837493863,"score_spread":0.1598127642725773,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}