{"id":"W4407730543","doi":"10.46690/elder.2024.04.01","title":"Capturing fine-grained teacher performance from student evaluation of teaching via ChatGPT","year":2024,"lang":"en","type":"article","venue":"Education and lifelong development research.","topic":"Online Learning and Analytics","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Beijing Normal University; Crohn's and Colitis Foundation of America","keywords":"Mathematics education; Computer science; Psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004429795,0.00009878418,0.0001060322,0.0003588494,0.0002401233,0.000255821,0.0003209885,0.00004312512,0.0000384005],"category_scores_gemma":[0.0001935653,0.00008564271,0.00002129976,0.0003217842,0.00003712506,0.0002941632,0.0001768852,0.0004175733,0.00003556195],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001522006,"about_ca_system_score_gemma":0.001224734,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001063804,"about_ca_topic_score_gemma":0.00001680489,"domain_scores_codex":[0.9978671,0.0002968407,0.0002537227,0.0003407934,0.00102094,0.000220596],"domain_scores_gemma":[0.9992525,0.0001360813,0.00004251791,0.0002325212,0.0002380026,0.0000984084],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000001946818,0.0001971231,0.01406859,0.00007189879,0.00005028082,8.007761e-7,0.02667571,0.00009881687,0.000255352,0.001939345,0.001397462,0.9552427],"study_design_scores_gemma":[0.0003800101,0.00007490536,0.3887188,0.00089533,0.0000269596,0.000006777089,0.002261644,0.5555422,0.0018342,0.002927043,0.04696688,0.0003653283],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9868639,0.00169992,0.00526718,0.003964572,0.0004381648,0.0001707653,3.439368e-7,0.0000777457,0.001517385],"genre_scores_gemma":[0.9830192,0.00003750323,0.01399618,0.00002893916,0.0002742193,0.00003404557,0.00002238779,0.000008919962,0.002578551],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9548774,"threshold_uncertainty_score":0.3492408,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05858031486306325,"score_gpt":0.4098003805682974,"score_spread":0.3512200657052341,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}