{"id":"W7132932718","doi":"","title":"Large scale performance-based assessment: Dentification of individual student gaps with implications for teacher content knowledge","year":2004,"lang":"","type":"dissertation","venue":"TSpace","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Nature versus nurture; Literacy; Population; Construct (python library); Scale (ratio); Pace; Knowledge acquisition; Task (project management); Interpretation (philosophy)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts"],"consensus_categories":[],"category_scores_codex":[0.001911966,0.0006545972,0.0008596467,0.0003970126,0.001541918,0.0003698136,0.001282937,0.0005170871,0.0004162898],"category_scores_gemma":[0.00003699282,0.0006249898,0.0003788842,0.001038553,0.0003663235,0.000357683,0.00008447329,0.0004775621,0.00003795382],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006943757,"about_ca_system_score_gemma":0.002664816,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001937006,"about_ca_topic_score_gemma":0.003260939,"domain_scores_codex":[0.9955102,0.0002479797,0.001003461,0.0009912665,0.001337327,0.0009097464],"domain_scores_gemma":[0.9956895,0.0002482074,0.001552668,0.0006623098,0.001592415,0.0002549044],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0005376168,0.009550278,0.5263824,0.001531376,0.001148945,5.128051e-7,0.4429919,0.0001498715,0.003559502,0.01028551,0.0009312186,0.002930926],"study_design_scores_gemma":[0.004031721,0.0006578234,0.7345037,0.0006107473,0.001133406,3.159686e-7,0.2550493,0.000139374,0.001067826,0.00004313917,0.002124496,0.0006381167],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.966263,0.0006511409,0.003705245,0.0007001699,0.0008684458,0.004904756,0.0001495187,0.00008374503,0.02267397],"genre_scores_gemma":[0.9666165,0.0002981952,0.001777773,0.00003115654,0.0003052871,0.001623671,0.002948693,0.00008915675,0.02630956],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2081213,"threshold_uncertainty_score":0.9997579,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06807253949039363,"score_gpt":0.436790008042412,"score_spread":0.3687174685520184,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}