{"id":"W4412712541","doi":"10.12968/eyed.2025.24.17.9","title":"Developing an assessment toolbox","year":2025,"lang":"en","type":"article","venue":"Early Years Educator","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Education and Early Childhood Development","funders":"","keywords":"Toolbox; Computer science; Medicine; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02369379,0.001333014,0.0008018876,0.003026701,0.001309153,0.005917279,0.002059137,0.001686519,0.01893764],"category_scores_gemma":[0.04591239,0.0007569924,0.001065369,0.00125254,0.001669345,0.008502351,0.007285004,0.00475724,0.01824526],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001803629,"about_ca_system_score_gemma":0.009994619,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001246618,"about_ca_topic_score_gemma":0.001997093,"domain_scores_codex":[0.9880579,0.006525651,0.001255049,0.0009053236,0.002851859,0.0004043506],"domain_scores_gemma":[0.9723328,0.01193373,0.001168165,0.002677602,0.008920358,0.002967238],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001298557,0.0009897741,0.005558498,0.0006794092,0.00004886095,0.0003082573,0.003933702,0.004764037,0.003668451,0.04694156,0.08129286,0.8516848],"study_design_scores_gemma":[0.0002137533,0.0008031672,0.008172831,0.004468948,0.00006971419,0.002041064,0.006027461,0.04349994,0.01147924,0.1444945,0.778386,0.000343381],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01168901,0.001071905,0.8931655,0.01026175,0.001249728,0.002644423,0.0007119869,0.01361638,0.06558928],"genre_scores_gemma":[0.03817145,0.0009779574,0.9387059,0.001194312,0.0001304216,0.002062702,0.0008973365,0.0008257838,0.01703415],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02369379,"threshold_uncertainty_score":0.1253062,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1874605579511776,"score_gpt":0.577382544379839,"score_spread":0.3899219864286615,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}