{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":14,"total_is_capped":false,"direct_labels_cover":0,"predictions_cover":14,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"9f297251ca0a","filters":{"venue":"Educational Evaluation and Policy Analysis"}},"results":[{"id":"W2130850661","doi":"10.3102/01623737027003205","title":"Effects of Kindergarten Retention Policy on Children’s Cognitive Growth in Reading and Mathematics","year":2005,"lang":"en","type":"article","venue":"Educational Evaluation and Policy Analysis","topic":"Early Childhood Education and Development","field":"Social Sciences","cited_by":277,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Grade retention; Psychology; Propensity score matching; Multilevel model; Limiting; Promotion (chess); Academic achievement; Developmental psychology; Homogeneous; Harm; Reading (process); Mathematics education; Longitudinal study; Cohort; Social psychology; Political science; Mathematics; Statistics","authors":[{"name":"Guanglei Hong","is_ca":true},{"name":"Stephen W. Raudenbush","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02019279729219308,"gpt":0.3748215988611014,"spread":0.3546288015689083,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009768583,0.0000874679,0.0001595764,0.001183847,0.0001934814,0.0000461345,0.00005118669,0.00006049834,0.0001463487],"category_scores_gemma":[0.002685388,0.00008961403,0.00005044956,0.001383102,0.00009881343,0.0001558698,0.00001137376,0.00006731094,0.000009872037],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002229111,"about_ca_system_score_gemma":0.001430043,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004493937,"about_ca_topic_score_gemma":0.0008610553,"domain_scores_codex":[0.9986736,0.0002644932,0.0002665679,0.0001881225,0.0004661868,0.0001409952],"domain_scores_gemma":[0.9988762,0.0005418973,0.0001634616,0.00005508628,0.0002589502,0.0001044617],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.00001786051,0.0008278751,0.2125872,0.00004388831,0.0006315956,6.393266e-8,0.1875346,0.0001955367,0.00003911583,0.5525708,0.0006255984,0.04492583],"study_design_scores_gemma":[0.0003665003,0.00001523711,0.9825962,0.00004871381,0.000210339,4.477389e-7,0.001545426,0.0003009549,0.0001272593,0.01463834,0.00005487278,0.00009566536],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9532658,0.0001944196,0.00006512259,0.03833193,0.00003506835,0.0003901583,0.000006446614,0.000009283595,0.007701718],"genre_scores_gemma":[0.9973404,0.0002725896,0.0005841904,0.0007209745,0.0004383422,0.00003727302,0.00006846745,0.000004792041,0.0005330158],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.770009,"threshold_uncertainty_score":0.6793518,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3008309614","doi":"10.3102/0162373720906217","title":"Teacher Coaching in a Simulated Environment","year":2020,"lang":"en","type":"article","venue":"Educational Evaluation and Policy Analysis","topic":"Teacher Education and Leadership Studies","field":"Social Sciences","cited_by":181,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Impact","funders":"National Academy of Education; University of Virginia; Spencer Foundation","keywords":"Coaching; Psychology; Perception; Mathematics education; Medical education; Teacher education; Pedagogy; Applied psychology; Medicine","authors":[{"name":"Julie Cohen","is_ca":false},{"name":"Vivian C. Wong","is_ca":false},{"name":"Anandita Krishnamachari","is_ca":false},{"name":"Rebekah Berlin","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.213634107861544,"gpt":0.4796332351777997,"spread":0.2659991273162557,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00110119,0.00007094062,0.0001241929,0.000286652,0.0001922311,0.00004690956,0.0000717614,0.00004463784,0.003232269],"category_scores_gemma":[0.001510746,0.00007497884,0.00005661916,0.001081945,0.00009763645,0.0001042205,0.00001213512,0.0001030285,0.00006201918],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000205149,"about_ca_system_score_gemma":0.0005086618,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004605237,"about_ca_topic_score_gemma":0.001118893,"domain_scores_codex":[0.9984828,0.0005955494,0.000181448,0.000198798,0.0003999477,0.0001414666],"domain_scores_gemma":[0.9994644,0.0002090263,0.00006727265,0.00006590744,0.00005368487,0.0001397349],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"observational","study_design_scores_codex":[0.000009566335,0.0001821689,0.2776143,0.00000532638,0.0002904202,6.464118e-8,0.6887025,0.008955454,0.00001200888,0.01551967,0.001592848,0.007115668],"study_design_scores_gemma":[0.000433478,0.00001247679,0.7834407,0.000004609107,0.0004987203,7.91478e-8,0.1414388,0.02540216,0.000001930832,0.00313585,0.045413,0.0002181602],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.271114,0.0004061492,0.00009359148,0.7208598,0.0000329317,0.0001874469,0.000002395602,0.00001637559,0.007287235],"genre_scores_gemma":[0.9948868,0.00007113071,0.0000988315,0.003556749,0.0005218919,0.00002211685,0.00004011486,0.000003823776,0.0007985095],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7237728,"threshold_uncertainty_score":0.9976789,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2156855199","doi":"10.3102/01623737022004357","title":"Power and Politics in the Adoption of School Reform Models","year":2000,"lang":"en","type":"article","venue":"Educational Evaluation and Policy Analysis","topic":"Educational Assessment and Improvement","field":"Decision Sciences","cited_by":174,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Education reform; Power (physics); Perspective (graphical); Politics; Process (computing); Public administration; Sustainability; Set (abstract data type); Political science; Public relations; Economic growth; Economics; Higher education","authors":[{"name":"Amanda Datnow","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1507326485546578,"gpt":0.4970251549872353,"spread":0.3462925064325775,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.003470145,0.00007908913,0.000148176,0.0007426881,0.0001171077,0.0001127883,0.0001760139,0.0000350368,0.004391372],"category_scores_gemma":[0.0005240427,0.0000517356,0.00006793295,0.001545968,0.00007681164,0.0003648571,0.00001471655,0.00006577664,0.00003318929],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001448912,"about_ca_system_score_gemma":0.0006006098,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009195157,"about_ca_topic_score_gemma":0.0001460981,"domain_scores_codex":[0.9976767,0.0002693984,0.0004950807,0.0002181314,0.001228353,0.0001123183],"domain_scores_gemma":[0.9986128,0.0004865942,0.0001471353,0.0002515222,0.0004305583,0.00007145171],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.00002588093,0.0005600401,0.07123416,0.000007285634,0.000235086,4.440465e-8,0.01045673,0.007152297,0.00005195538,0.8724257,0.003659745,0.03419104],"study_design_scores_gemma":[0.0001590738,0.0000230046,0.6872438,0.000002921161,0.0001035493,9.491096e-7,0.002757199,0.02301989,0.000003392797,0.2850631,0.001566708,0.00005651563],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9124329,0.0002161416,0.00007056484,0.05434098,0.00003065908,0.0001792052,0.00001613711,0.000001767463,0.03271164],"genre_scores_gemma":[0.9951637,0.00008668088,0.0003222642,0.001396429,0.0001311354,0.00004322462,0.00006431201,0.00000231335,0.002789975],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6160096,"threshold_uncertainty_score":0.9965187,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2124978570","doi":"10.3102/0162373707309073","title":"Early-Grade Retention and Children’s Reading and Math Learning in Elementary Years","year":2007,"lang":"en","type":"article","venue":"Educational Evaluation and Policy Analysis","topic":"Early Childhood Education and Development","field":"Social Sciences","cited_by":99,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Grade retention; Psychology; Reading (process); Retention rate; Mathematics education; Early childhood; Intervention (counseling); Developmental psychology; Academic achievement; Computer science","authors":[{"name":"Guanglei Hong","is_ca":true},{"name":"Bing Yu","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02636358941870153,"gpt":0.382660152524002,"spread":0.3562965631053004,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00284517,0.00006946898,0.0001022664,0.0007511431,0.0003162707,0.0001014485,0.00003724804,0.00005023063,0.0001939527],"category_scores_gemma":[0.0003860968,0.00007890243,0.00002738783,0.0009069371,0.00008326621,0.0001870892,0.00001653498,0.00009714829,0.000005375621],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001299195,"about_ca_system_score_gemma":0.0004505508,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01095705,"about_ca_topic_score_gemma":0.003393594,"domain_scores_codex":[0.9988192,0.0001996419,0.0002247247,0.0002115057,0.000374239,0.000170679],"domain_scores_gemma":[0.9994901,0.0001520998,0.00009430854,0.0000531229,0.00007976087,0.0001306317],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000003731328,0.00003206977,0.9102052,0.000001866361,0.00008310622,6.474436e-8,0.03605857,0.00002741686,0.00001134144,0.02755443,0.00008960489,0.02593262],"study_design_scores_gemma":[0.0001876824,0.000008122168,0.9918594,0.00000806247,0.0001088403,8.755887e-7,0.003659945,0.0001849934,0.000001955191,0.002935336,0.0009596765,0.00008507036],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9868891,0.000254304,0.00003767754,0.01075889,0.00004791789,0.0001677014,0.000001660039,0.00001172905,0.001831013],"genre_scores_gemma":[0.997331,0.0003919049,0.0007083379,0.0003973994,0.0002538846,0.00001123697,0.00009199524,0.000004131229,0.0008100789],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.08165426,"threshold_uncertainty_score":0.9956291,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2144062847","doi":"10.3102/01623737024003201","title":"Do Low-Achieving Students Benefit More from Small Classes? Evidence from the Tennessee Class Size Experiment","year":2002,"lang":"en","type":"article","venue":"Educational Evaluation and Policy Analysis","topic":"School Choice and Performance","field":"Social Sciences","cited_by":96,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Class size; Differential effects; Class (philosophy); Mathematics education; Academic achievement; Randomized experiment; Psychology; Reading (process); Differential (mechanical device); Student achievement; Scale (ratio); Quarter (Canadian coin); Mathematics; Statistics; Computer science; Physics; Geography; Political science","authors":[{"name":"Barbara Nye","is_ca":false},{"name":"Larry V. Hedges","is_ca":false},{"name":"Spyros Konstantopoulos","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1088547008286126,"gpt":0.4471015832571122,"spread":0.3382468824284997,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001296953,0.0001453503,0.0001890328,0.0001338121,0.0009100875,0.0004820907,0.0005064966,0.00008486442,0.01007686],"category_scores_gemma":[0.002092099,0.0001166048,0.0001343902,0.00119324,0.0001426256,0.000403944,0.00006102394,0.0001400232,0.0001297476],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000257907,"about_ca_system_score_gemma":0.0003217219,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03343158,"about_ca_topic_score_gemma":0.01429918,"domain_scores_codex":[0.997378,0.0003875312,0.0003079117,0.000367999,0.00132825,0.000230316],"domain_scores_gemma":[0.996651,0.002314949,0.0001978677,0.0003421572,0.0003373812,0.0001566221],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00001676457,0.0002071643,0.9226873,0.000003608599,0.0006804217,1.494412e-7,0.0618125,0.001075859,0.0001806825,0.001415221,0.00431313,0.007607205],"study_design_scores_gemma":[0.0002063069,0.000008517618,0.9804381,0.0000524497,0.0006770285,6.973824e-8,0.007373558,0.004532584,0.00002791942,0.001502805,0.005010586,0.0001701431],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9181305,0.003778516,0.00001820104,0.07637128,0.0001894343,0.0002150991,0.00003395345,0.00001559138,0.001247427],"genre_scores_gemma":[0.9908316,0.001529563,0.0001975118,0.00285374,0.002712733,0.0001244862,0.00006800943,0.000007315917,0.001675114],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.07351755,"threshold_uncertainty_score":0.990828,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1980838979","doi":"10.3102/01623737027001001","title":"Success for All: First-Year Results From the National Randomized Field Trial","year":2005,"lang":"en","type":"article","venue":"Educational Evaluation and Policy Analysis","topic":"School Choice and Performance","field":"Social Sciences","cited_by":84,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Randomized experiment; Quarter (Canadian coin); Test (biology); Psychology; Randomized controlled trial; Multilevel model; Control (management); Treatment and control groups; Reading (process); Mathematics education; Statistics; Medicine; Mathematics; Computer science; Political science; Geography","authors":[{"name":"Geoffrey D. Borman","is_ca":false},{"name":"Robert E. Slavin","is_ca":false},{"name":"Alan Cheung","is_ca":false},{"name":"Anne Chamberlain","is_ca":false},{"name":"Nancy A. Madden","is_ca":false},{"name":"Bette Chambers","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.09148252164579562,"gpt":0.4777966616051301,"spread":0.3863141399593344,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00530625,0.00007420599,0.0001773031,0.000166956,0.0006722764,0.0001490899,0.0001724478,0.00006946439,0.001629785],"category_scores_gemma":[0.01363002,0.00005553096,0.0001946219,0.0006047653,0.0001004307,0.0002461806,0.00001036951,0.00006821372,0.00003431012],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001153415,"about_ca_system_score_gemma":0.0009487715,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.009939523,"about_ca_topic_score_gemma":0.0262929,"domain_scores_codex":[0.9981626,0.0004112064,0.0003305115,0.000194459,0.0007644497,0.000136737],"domain_scores_gemma":[0.9920203,0.007086719,0.0001770596,0.0001015826,0.0005418729,0.00007243117],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.2049459,0.0004038739,0.01097264,0.000007214694,0.003375257,1.591555e-8,0.04875922,0.009509581,0.000002678702,0.246451,0.4597055,0.01586712],"study_design_scores_gemma":[0.3219653,0.00005460146,0.02213892,0.00001395347,0.002718351,9.256178e-8,0.001500741,0.05194182,0.00001817175,0.06947659,0.5298158,0.0003556511],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.194245,0.0002354526,0.0002732514,0.7853158,0.0003037759,0.00112145,0.0001869348,0.0000168593,0.01830158],"genre_scores_gemma":[0.9798663,0.0002265717,0.000581775,0.006394526,0.01006703,0.000348274,0.0004903023,0.000003532216,0.002021678],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7856213,"threshold_uncertainty_score":0.9992828,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2153308398","doi":"10.3102/01623737023003251","title":"Class Size and Eighth-Grade Math Achievement in the United States and Abroad","year":2001,"lang":"en","type":"article","venue":"Educational Evaluation and Policy Analysis","topic":"School Choice and Performance","field":"Social Sciences","cited_by":82,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Class size; Mathematics education; Curriculum; Class (philosophy); Student achievement; Academic achievement; Math education; Psychology; Pedagogy; Computer science","authors":[{"name":"Suet‐ling Pong","is_ca":false},{"name":"Aaron M. Pallas","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05050335955880637,"gpt":0.4362833388378362,"spread":0.3857799792790298,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001679092,0.00006505414,0.00008553884,0.0003208914,0.0003847465,0.000129574,0.00007135087,0.00003571987,0.0002974498],"category_scores_gemma":[0.0003644263,0.00004903362,0.00002257856,0.001564498,0.000116082,0.0001598573,0.00001041358,0.00007100085,0.000004988549],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000560644,"about_ca_system_score_gemma":0.0002213735,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.01904708,"about_ca_topic_score_gemma":0.01879295,"domain_scores_codex":[0.9988443,0.0003444011,0.0001447561,0.0001376789,0.0003993803,0.0001294455],"domain_scores_gemma":[0.9991557,0.0005346884,0.00006073173,0.00008253274,0.00009881904,0.00006755573],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0000186381,0.0001549827,0.8749432,0.000008317875,0.0001642082,1.846888e-7,0.05833197,0.000678323,0.000009743751,0.05839717,0.002544132,0.004749131],"study_design_scores_gemma":[0.0002007758,0.00001065644,0.9204629,0.000003954623,0.000173555,5.649806e-7,0.007261671,0.008256371,6.589066e-7,0.009025307,0.05453733,0.00006623167],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8555945,0.0003316723,0.000004872946,0.1421307,0.00001901485,0.0001138557,0.000005027587,0.000003802088,0.00179651],"genre_scores_gemma":[0.9906656,0.003980331,0.00005659995,0.004002596,0.0003328701,0.00003902511,0.00009283672,0.00000206073,0.0008281364],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1381281,"threshold_uncertainty_score":0.9991115,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2146700645","doi":"10.3102/0162373708328259","title":"Reading Instruction Time and Homogeneous Grouping in Kindergarten: An Application of Marginal Mean Weighting Through Stratification","year":2008,"lang":"en","type":"article","venue":"Educational Evaluation and Policy Analysis","topic":"School Choice and Performance","field":"Social Sciences","cited_by":58,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Reading (process); Psychology; Homogeneous; Weighting; Mathematics education; Multilevel model; Reading comprehension; Longitudinal study; Set (abstract data type); Computer science; Statistics; Linguistics; Mathematics","authors":[{"name":"Guanglei Hong","is_ca":true},{"name":"Yihua Hong","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04578867278231784,"gpt":0.3879297195275477,"spread":0.3421410467452298,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001145079,0.00007770482,0.0001408949,0.0004475833,0.000479664,0.00003563631,0.00007055579,0.00006957082,0.000137542],"category_scores_gemma":[0.0001919159,0.00008636519,0.00003293436,0.001372934,0.000140598,0.0006421753,0.000007519374,0.00007140198,0.000005372522],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001275026,"about_ca_system_score_gemma":0.0004345933,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009075721,"about_ca_topic_score_gemma":0.005806707,"domain_scores_codex":[0.9986547,0.0002648179,0.0002956385,0.0002324342,0.0004205433,0.0001319136],"domain_scores_gemma":[0.9992295,0.0001185781,0.0002123456,0.000125852,0.0002487337,0.00006497536],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00005081475,0.0002934352,0.7063082,0.00004068387,0.0002246965,2.413622e-7,0.112269,0.007030978,0.003689171,0.1040839,0.0001207226,0.06588816],"study_design_scores_gemma":[0.0003672708,0.00002873548,0.9024457,0.00001470988,0.0002402915,0.000006078698,0.003280442,0.08054835,0.000198178,0.01206397,0.0006226008,0.0001836459],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9924045,0.0001153011,0.0004214697,0.003967687,0.00002826937,0.0001814088,0.000004467255,0.00001060741,0.002866334],"genre_scores_gemma":[0.9976699,0.0002880788,0.001111687,0.0001091917,0.0004808957,0.0000366134,0.0001757273,0.000004367832,0.0001235271],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1961375,"threshold_uncertainty_score":0.997523,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1987197661","doi":"10.3102/0162373711424206","title":"Differential Effects of Literacy Instruction Time and Homogeneous Ability Grouping in Kindergarten Classrooms","year":2011,"lang":"en","type":"article","venue":"Educational Evaluation and Policy Analysis","topic":"School Choice and Performance","field":"Social Sciences","cited_by":58,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Psychology; Homogeneous; Literacy; Mathematics education; Developmental psychology; Early literacy; Cognition; Cognitive development; Differential effects; Longitudinal study; Early childhood; Pedagogy; Statistics","authors":[{"name":"Guanglei Hong","is_ca":false},{"name":"Carl Corter","is_ca":true},{"name":"Yihua Hong","is_ca":true},{"name":"Janette Pelletier","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02560448217574205,"gpt":0.3647464052251619,"spread":0.3391419230494199,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0004885718,0.00006405917,0.0001379524,0.000443701,0.0001578032,0.00003053963,0.00005543554,0.0000590759,0.001088976],"category_scores_gemma":[0.0004882748,0.00006413607,0.00004764918,0.0008498905,0.0001153241,0.0002716321,0.00001281793,0.00005895468,0.000005435942],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007685035,"about_ca_system_score_gemma":0.0002684072,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007503607,"about_ca_topic_score_gemma":0.003263162,"domain_scores_codex":[0.9989434,0.000291478,0.0002044612,0.0001611713,0.0002896361,0.0001098431],"domain_scores_gemma":[0.9993614,0.0002062322,0.0001086842,0.00009087933,0.0001647121,0.00006807493],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00004617987,0.000231659,0.8623029,0.00005508895,0.0002111526,1.134706e-7,0.05803106,0.00007081983,0.0004644388,0.005949047,0.00004392822,0.07259363],"study_design_scores_gemma":[0.0002713807,0.00001732432,0.9876843,0.00001216773,0.0002349577,3.563213e-7,0.000258493,0.004114683,0.00008805975,0.007143195,0.0001043561,0.00007069157],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.996866,0.0001331572,0.00003683785,0.001099859,0.00007749071,0.0001422176,0.000003179932,0.000005136266,0.001636133],"genre_scores_gemma":[0.9991299,0.0001221058,0.0001493415,0.00006725391,0.0003063313,0.00002289168,0.00003137564,0.000002440848,0.0001683719],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1253815,"threshold_uncertainty_score":0.9998242,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2791023047","doi":"10.3102/0162373718760218","title":"School Improvement Grants in Ohio: Effects on Student Achievement and School Administration","year":2018,"lang":"en","type":"article","venue":"Educational Evaluation and Policy Analysis","topic":"School Choice and Performance","field":"Social Sciences","cited_by":47,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Regression discontinuity design; Quarter (Canadian coin); Psychology; Demographic economics; Turnover; Regression analysis; Mathematics education; Economics; Statistics; Geography; Mathematics; Management","authors":[{"name":"Deven Carlson","is_ca":false},{"name":"Stéphane Lavertu","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03287081499345695,"gpt":0.4489629340119663,"spread":0.4160921190185093,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.00178704,0.0001055591,0.0001388185,0.0005617331,0.0004479781,0.0001509513,0.00009248906,0.00005908761,0.001054875],"category_scores_gemma":[0.000869147,0.0001019558,0.00004136054,0.001026866,0.00009159167,0.0002444823,0.00001811028,0.00009622252,0.0001134215],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003135856,"about_ca_system_score_gemma":0.00102072,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002092679,"about_ca_topic_score_gemma":0.016002,"domain_scores_codex":[0.9982662,0.0002744882,0.0002469619,0.0002723778,0.0007475344,0.0001925011],"domain_scores_gemma":[0.9991215,0.0001817311,0.0001074506,0.000139029,0.0002248626,0.0002254298],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00007795193,0.0004756957,0.9434243,0.00002050597,0.0003091417,1.923671e-7,0.007129973,0.00005523336,0.0002268076,0.0149333,0.002063009,0.03128389],"study_design_scores_gemma":[0.0004804218,0.0001892556,0.9925522,0.00001940801,0.0001938574,8.351221e-8,0.0008102916,0.000447968,0.000146081,0.002290417,0.002757338,0.0001126597],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9646601,0.000105073,0.00001135545,0.02891368,0.0001405039,0.0004156332,0.000004485418,0.000008570862,0.005740606],"genre_scores_gemma":[0.9944037,0.0001921951,0.00007067261,0.002330943,0.001662367,0.0001529084,0.00004336508,0.00000398487,0.001139901],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04912793,"threshold_uncertainty_score":0.9998583,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1977534634","doi":"10.3102/0162373711398125","title":"Who Benefits From Kindergarten? Evidence From the Introduction of State Subsidization","year":2011,"lang":"en","type":"article","venue":"Educational Evaluation and Policy Analysis","topic":"Early Childhood Education and Development","field":"Social Sciences","cited_by":39,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Socioeconomic status; Subsidy; Immigration; Demographic economics; Revenue; Psychology; State (computer science); Mathematics education; Economic growth; Economics; Demography; Political science; Sociology; Population; Computer science","authors":[{"name":"Elizabeth Dhuey","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08998701385613952,"gpt":0.3771080661541513,"spread":0.2871210522980118,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001407931,0.00008770316,0.0001406053,0.0002490703,0.0004012878,0.00006499777,0.0001615165,0.00004929932,0.006245437],"category_scores_gemma":[0.002602352,0.00007437931,0.00006941144,0.00147666,0.0001471604,0.0003384135,0.0000185165,0.00006499718,0.00003115092],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001272065,"about_ca_system_score_gemma":0.00217119,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.07333943,"about_ca_topic_score_gemma":0.01746355,"domain_scores_codex":[0.9980599,0.0005511189,0.0003234825,0.0002562781,0.0006845559,0.0001246834],"domain_scores_gemma":[0.9981301,0.0005173509,0.0002663494,0.0002204685,0.0007492099,0.000116454],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00003095227,0.0001742228,0.5034853,0.000002683899,0.0008652936,1.68632e-8,0.4083047,0.00104352,0.0000539504,0.04147931,0.01050707,0.03405301],"study_design_scores_gemma":[0.00008429023,0.000006544087,0.9708068,0.00001185065,0.0004054239,4.383734e-8,0.003536251,0.0003467139,0.0001317645,0.02315172,0.001435017,0.00008363093],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9191489,0.001188993,0.0003956201,0.07706423,0.0002902206,0.0002617702,0.00004137852,0.00001438299,0.001594517],"genre_scores_gemma":[0.9955035,0.0008475981,0.0007942874,0.0006670085,0.001277168,0.00002869323,0.0002870333,0.000004826171,0.0005899102],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4673214,"threshold_uncertainty_score":0.994663,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2804370236","doi":"10.3102/0162373718782648","title":"Course Choice and Achievement Effects of a System-Wide Vocational Education and Training Voucher Scheme for Young People","year":2018,"lang":"en","type":"article","venue":"Educational Evaluation and Policy Analysis","topic":"Education Systems and Policy","field":"Social Sciences","cited_by":8,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Queen's University; National Centre for Vocational Education Research; Queen's University Belfast","keywords":"Voucher; Vocational education; Equity (law); School choice; Academic achievement; Training (meteorology); Quality (philosophy); Public economics; Psychology; Mathematics education; Economic growth; Medical education; Pedagogy; Political science; Business; Economics; Accounting","authors":[{"name":"Duncan McVicar","is_ca":false},{"name":"Cain Polidano","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04401642896527869,"gpt":0.4417211995699809,"spread":0.3977047706047022,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001859078,0.0001208801,0.0002383598,0.0004547114,0.0006137711,0.00009667996,0.00007859156,0.0000825945,0.0000828384],"category_scores_gemma":[0.001695037,0.0001253988,0.00006548029,0.0009111557,0.0002321113,0.0002076175,0.00001530148,0.00004431824,0.000003489605],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001656052,"about_ca_system_score_gemma":0.002989915,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0116907,"about_ca_topic_score_gemma":0.005219196,"domain_scores_codex":[0.9984038,0.0003267427,0.0003470185,0.0002825025,0.0004540184,0.0001859247],"domain_scores_gemma":[0.9973288,0.0009897075,0.0003041101,0.0001432391,0.001045734,0.0001884345],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.0000129967,0.0003557332,0.3725101,0.0004093351,0.0007475663,4.471052e-9,0.1784608,0.00001197209,0.000139689,0.4259551,0.004159544,0.01723711],"study_design_scores_gemma":[0.0004278323,0.00004658497,0.9583454,0.00008468175,0.0009197279,0.000001458048,0.0241695,0.002710286,0.00001612653,0.002979078,0.01012452,0.0001748695],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9647971,0.00102514,0.000948449,0.02611509,0.0007125583,0.0009976419,0.00002874122,0.00001790455,0.005357371],"genre_scores_gemma":[0.9923115,0.00004138058,0.002239264,0.0003800159,0.002817455,0.0004307654,0.0001229999,0.000008871996,0.001647757],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5858352,"threshold_uncertainty_score":0.9948905,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4390877660","doi":"10.3102/01623737231218735","title":"Beyond Prescriptive Reforms: An Examination of North Carolina’s Flexible School Restart Program","year":2024,"lang":"en","type":"article","venue":"Educational Evaluation and Policy Analysis","topic":"School Choice and Performance","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Autonomy; Flexibility (engineering); Principal (computer security); Quarter (Canadian coin); School choice; Mathematics education; South carolina; Psychology; Pedagogy; Public administration; Sociology; Political science; Economics; Management; Computer science; Law; History","authors":[{"name":"Lam Pham","is_ca":false},{"name":"Gage F. Matthews","is_ca":false},{"name":"Timothy A. Drake","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05610033606833933,"gpt":0.4434867261868464,"spread":0.3873863901185071,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001904466,0.00009157133,0.0001371352,0.0008180619,0.0003137992,0.0001356968,0.0001248592,0.00006135531,0.0008482287],"category_scores_gemma":[0.0006361458,0.00008404064,0.00008427526,0.002576338,0.0001408848,0.0009206339,0.00001280574,0.0001069914,0.00003093227],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002519514,"about_ca_system_score_gemma":0.002081132,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.008128035,"about_ca_topic_score_gemma":0.03119427,"domain_scores_codex":[0.9982144,0.0003388451,0.0002655057,0.0002517627,0.0007621443,0.0001673006],"domain_scores_gemma":[0.9988592,0.000114869,0.00009379089,0.000168976,0.0006187853,0.0001443761],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00004458128,0.0005083465,0.3255979,0.0001348283,0.001039098,2.840929e-7,0.06759626,0.002424709,0.00003681812,0.1362668,0.005675547,0.4606749],"study_design_scores_gemma":[0.00008782254,0.00006163336,0.9681155,0.00001491965,0.0005293024,2.299233e-7,0.001450266,0.01117823,0.00001824493,0.00730716,0.01113055,0.0001061254],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9642903,0.0006675699,0.00002357276,0.007987629,0.0001672792,0.0003907196,0.00003562251,0.00005633948,0.02638098],"genre_scores_gemma":[0.9933253,0.0002913464,0.0004652353,0.0001774135,0.001321167,0.0001631189,0.0004211224,0.000006319679,0.003828983],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6425176,"threshold_uncertainty_score":0.9984769,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4407384694","doi":"10.3102/01623737241311537","title":"Measuring Grading Standards at High Schools: A Methodological Contribution, an Example, and Some Policy Implications","year":2025,"lang":"en","type":"article","venue":"Educational Evaluation and Policy Analysis","topic":"Higher Education Learning Practices","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true},"ca_institutions":"Wilfrid Laurier University","funders":"Canadian Institutes of Health Research; University of Alberta","keywords":"Grading (engineering); Academic standards; Mathematics education; Policy analysis; Higher education; Econometrics; Psychology; Political science; Economics; Economic growth; Public administration; Engineering","authors":[],"retraction":null,"screen_n_in":null,"score":{"opus":0.2785140261361238,"gpt":0.5528497145082416,"spread":0.2743356883721178,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","sts"],"consensus_categories":[],"category_scores_codex":[0.009614435,0.0001470931,0.0002774922,0.001118429,0.002131317,0.0003595375,0.000181708,0.0001298525,0.0007471397],"category_scores_gemma":[0.01825696,0.0001524565,0.00008135653,0.003160891,0.000270999,0.0007671216,0.00005666643,0.0001836117,0.000010272],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00144054,"about_ca_system_score_gemma":0.005138886,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03066875,"about_ca_topic_score_gemma":0.003612088,"domain_scores_codex":[0.9949667,0.003081406,0.0003901523,0.0004906548,0.000770348,0.0003007132],"domain_scores_gemma":[0.995616,0.001942897,0.0002857046,0.0002911877,0.001564486,0.0002996946],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.00001649671,0.00008736865,0.08591696,0.00000659839,0.0003608128,2.21452e-8,0.007365124,0.0004106984,0.000103292,0.8992026,0.001135398,0.005394621],"study_design_scores_gemma":[0.0003155639,0.00001507169,0.7717223,0.00001124736,0.0008693979,0.000001100308,0.002033581,0.0003895997,0.00001784639,0.1487566,0.07569472,0.0001729887],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6317106,0.0009176367,0.002753463,0.3606406,0.000173709,0.0004460443,0.00005506771,0.00007038638,0.003232522],"genre_scores_gemma":[0.9913523,0.0004064797,0.002550978,0.001753022,0.0008895954,0.0001999677,0.000219902,0.000006980098,0.002620752],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.750446,"threshold_uncertainty_score":0.9991678,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}