{"id":"W3034035587","doi":"10.31045/jes.3.2.7","title":"A Thirty State Analysis of Teacher Supervision and Evaluation Systems in the ESSA Era","year":2020,"lang":"en","type":"article","venue":"Journal of Educational Supervision","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"University of Toronto; University of Minnesota; Princeton University","keywords":"Formative assessment; Accountability; Summative assessment; Public administration; Legislation; Policy analysis; Politics; State (computer science); Political science; Psychology; Law; Pedagogy; Computer science","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.01379469,0.0001113166,0.000398558,0.000686288,0.00009206853,0.0001863878,0.0005171828,0.00005113576,0.002005663],"category_scores_gemma":[0.002363423,0.00006132821,0.0001638089,0.002211273,0.00004686413,0.0007428056,0.00005014347,0.0002511623,0.00001550619],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006950712,"about_ca_system_score_gemma":0.0006141438,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008073195,"about_ca_topic_score_gemma":0.00005424644,"domain_scores_codex":[0.9936272,0.001121001,0.001387395,0.0002157611,0.00353189,0.0001167906],"domain_scores_gemma":[0.9955841,0.001772795,0.000655313,0.000244986,0.001634637,0.0001081542],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0002703677,0.0006724861,0.7803757,0.00003671369,0.000325216,0.000001774412,0.04626544,0.07935133,0.001731823,0.002414897,0.009046923,0.07950731],"study_design_scores_gemma":[0.0005066491,0.0001814181,0.7308679,0.00003729722,0.0001965028,0.000007059778,0.007652014,0.2566926,0.00001515667,0.001886859,0.001887764,0.00006882809],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9726123,0.001703229,0.000178,0.02440632,0.0002433206,0.0002185819,0.000009049795,0.000001114382,0.0006281268],"genre_scores_gemma":[0.9987528,0.0001944669,0.0003277376,0.0005050346,0.0001464935,0.000006390292,0.00001501411,0.000004408842,0.00004768095],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1773413,"threshold_uncertainty_score":0.9989066,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1739923189409111,"score_gpt":0.4769084655628433,"score_spread":0.3029161466219322,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}