{"id":"W4410519181","doi":"10.1002/tea.70009","title":"A Multimodal Interactive Framework for Science Assessment in the Era of Generative Artificial Intelligence","year":2025,"lang":"en","type":"article","venue":"Journal of Research in Science Teaching","topic":"Educational Games and Gamification","field":"Psychology","cited_by":13,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Institute of Education Sciences; University of Georgia; U.S. Department of Education","keywords":"Generative grammar; Science education; Mathematics education; Computer science; Cognitive science; Artificial intelligence; Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.007073434,0.0008905487,0.0004161786,0.003444372,0.001690723,0.006216067,0.001818917,0.001664949,0.006168755],"category_scores_gemma":[0.01139317,0.0003996114,0.0008375644,0.001013009,0.0107281,0.005090952,0.006137615,0.002041418,0.000530144],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002395566,"about_ca_system_score_gemma":0.001553623,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002231929,"about_ca_topic_score_gemma":0.002901579,"domain_scores_codex":[0.9948901,0.003965249,0.000158335,0.0002917882,0.0005039105,0.0001905532],"domain_scores_gemma":[0.9925097,0.005719052,0.0002795328,0.0005431405,0.0005157989,0.0004327651],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009472777,0.0001392958,0.002425074,0.000400481,0.0000307383,0.0005450144,0.02868301,0.01259441,0.004752837,0.8751219,0.001564011,0.0736485],"study_design_scores_gemma":[0.00005633516,0.000179865,0.002662388,0.001023503,0.00005971758,0.0006671219,0.01318971,0.1066497,0.003423052,0.7971255,0.07485569,0.0001073959],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03833291,0.0006162576,0.9113194,0.003409379,0.00008216282,0.0003149139,0.0001093661,0.0008687224,0.04494683],"genre_scores_gemma":[0.588577,0.0003275133,0.4072931,0.0002708826,0.00003593745,0.0005311042,0.00007309326,0.0001015258,0.002789923],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9929265,"threshold_uncertainty_score":0.03740835,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1873675951264364,"score_gpt":0.595336900515216,"score_spread":0.4079693053887796,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}