{"id":"W4399782080","doi":"10.3390/ai5020045","title":"AI Detection of Human Understanding in a Gen-AI Tutor","year":2024,"lang":"en","type":"article","venue":"AI","topic":"Intelligent Tutoring Systems and Adaptive Learning","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"TUTOR; Computer science; Psychology; Artificial intelligence; Mathematics education","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008930571,0.0003181609,0.0003127813,0.0004121778,0.0002535412,0.001563978,0.000453633,0.0005606854,0.003506901],"category_scores_gemma":[0.005628162,0.0001213824,0.0001803281,0.0001392079,0.0005578463,0.001087098,0.001243192,0.0004604213,0.0009603588],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003250786,"about_ca_system_score_gemma":0.0002587484,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003703127,"about_ca_topic_score_gemma":0.0005044626,"domain_scores_codex":[0.9993543,0.0002261271,0.00003265916,0.0001634396,0.0001773474,0.00004606479],"domain_scores_gemma":[0.9986801,0.0007152987,0.000135438,0.00009243354,0.0002254882,0.0001512519],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001385808,0.000915162,0.1069745,0.000506774,0.0001024876,0.001009874,0.02053841,0.007544602,0.4495097,0.005948829,0.004189495,0.4013745],"study_design_scores_gemma":[0.0001699426,0.004493922,0.2501083,0.0001887042,0.0002008926,0.002308676,0.01181494,0.3646704,0.3038052,0.01575182,0.046135,0.0003521913],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.881137,0.0001549214,0.1048282,0.0003569984,0.00005929951,0.00018257,0.0001624499,0.00307326,0.01004522],"genre_scores_gemma":[0.9595769,0.00006418952,0.03602261,0.0001382973,0.00001409417,0.0001029917,0.0001166185,0.00007149135,0.003892848],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003506901,"threshold_uncertainty_score":0.01173174,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04571073162187732,"score_gpt":0.300980527689497,"score_spread":0.2552697960676196,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}