{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":6,"total_is_capped":false,"direct_labels_cover":0,"predictions_cover":6,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"94626412b82e","filters":{"venue":"Educational Assessment"}},"results":[{"id":"W2523280788","doi":"10.1080/10627197.2016.1236677","title":"Approaches to Classroom Assessment Inventory: A New Instrument to Support Teacher Assessment Literacy","year":2016,"lang":"en","type":"article","venue":"Educational Assessment","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":112,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"","keywords":"Accountability; Literacy; Construct (python library); Educational assessment; Psychology; Construct validity; Mathematics education; Standards for Educational and Psychological Testing; Standardized test; Standards-based assessment; Process (computing); Authentic assessment; Pedagogy; Psychometrics; Higher education; Computer science; Political science; Education theory; Curriculum","authors":[{"name":"Christopher DeLuca","is_ca":true},{"name":"Danielle LaPointe-McEwan","is_ca":true},{"name":"Ulemu Luhanga","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1328510477980371,"gpt":0.4138136676963231,"spread":0.280962619898286,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008253588,0.000430441,0.0005822259,0.003858602,0.0009327062,0.001996543,0.001059876,0.0004075276,0.003060797],"category_scores_gemma":[0.03430567,0.000440701,0.0006528134,0.002093418,0.0008376984,0.003129363,0.003665721,0.002538804,0.001466707],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001490116,"about_ca_system_score_gemma":0.005966381,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003421281,"about_ca_topic_score_gemma":0.009186864,"domain_scores_codex":[0.9914857,0.003093509,0.001549962,0.000520675,0.00306019,0.000290035],"domain_scores_gemma":[0.9712158,0.01304952,0.004314045,0.002022656,0.007819675,0.001578296],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0001697582,0.001226021,0.1660303,0.0004491439,0.00008644386,0.0001909056,0.01123625,0.0008948211,0.004672242,0.006215385,0.02711555,0.7817132],"study_design_scores_gemma":[0.0002605428,0.001775181,0.6868221,0.0009772708,0.0002240202,0.001873028,0.01090519,0.01124982,0.006550833,0.01668364,0.2623265,0.0003519336],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5186203,0.003280367,0.3064511,0.008135222,0.001254906,0.01422834,0.01033379,0.009342569,0.1283535],"genre_scores_gemma":[0.3423016,0.001643281,0.625768,0.0009850942,0.0002526271,0.01355159,0.003683569,0.0005600009,0.01125427],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008253588,"threshold_uncertainty_score":0.04364967,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1999929076","doi":"10.1080/10627197.2011.584042","title":"Voices From Test-Takers: Further Evidence for Language Assessment Validation and Use","year":2011,"lang":"en","type":"article","venue":"Educational Assessment","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":56,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"Guangdong University of Foreign Studies; Ministry of Education, India; Ministry of Earth Sciences","keywords":"Test (biology); Language assessment; Psychology; Test validity; Scale (ratio); Coding (social sciences); Test score; Computer science; Psychometrics; Mathematics education; Standardized test; Developmental psychology","authors":[{"name":"Liying Cheng","is_ca":true},{"name":"Christopher DeLuca","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1441623853519503,"gpt":0.4552257976525222,"spread":0.3110634123005719,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1679558,0.0006862718,0.001460696,0.003745166,0.005046748,0.007357724,0.003083837,0.0028186,0.002667543],"category_scores_gemma":[0.5089094,0.0008900915,0.001389184,0.002010464,0.009811517,0.00712669,0.01106563,0.004901371,0.0006093996],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002931892,"about_ca_system_score_gemma":0.003802686,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004104786,"about_ca_topic_score_gemma":0.00371341,"domain_scores_codex":[0.6754895,0.2317346,0.02039349,0.008941623,0.05831548,0.005125324],"domain_scores_gemma":[0.2965513,0.5815055,0.04345421,0.02513368,0.04700278,0.006352395],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0003438988,0.0001300095,0.07841828,0.0005830973,0.0001187508,0.000871143,0.8657131,0.000051526,0.002517341,0.00100734,0.0007533315,0.04949212],"study_design_scores_gemma":[0.0001141921,0.00116446,0.1429631,0.002713463,0.0001657072,0.003143332,0.8113228,0.001001511,0.007190667,0.003345808,0.02659477,0.0002802153],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9781256,0.002971441,0.006102022,0.007367013,0.0001826752,0.0001152881,0.0001072719,0.00004695568,0.004981703],"genre_scores_gemma":[0.9945524,0.0007247289,0.00183059,0.001846136,0.0001106038,0.0001243077,0.00008408133,0.00006191035,0.0006652317],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1679558,"threshold_uncertainty_score":0.8882459,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2975545495","doi":"10.1080/10627197.2019.1670056","title":"Toward a Teacher Professional Learning Continuum in Assessment for Learning","year":2019,"lang":"en","type":"article","venue":"Educational Assessment","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":52,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Council of Ministers of Education; Queen's University","funders":"","keywords":"Mathematics education; Assessment for learning; Professional learning community; Psychology; Pedagogy; Professional development; Observational study; Empirical research; Epistemology; Mathematics; Formative assessment","authors":[{"name":"Christopher DeLuca","is_ca":true},{"name":"Allison E. A. Chapman-Chin","is_ca":true},{"name":"Don A. Klinger","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03380106385401103,"gpt":0.4305736501055339,"spread":0.3967725862515229,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01738891,0.0001627612,0.0002504753,0.001745799,0.004056455,0.008004634,0.001155369,0.001695304,0.001664006],"category_scores_gemma":[0.03235021,0.0005196913,0.0002965005,0.001102683,0.007727739,0.006788302,0.008038673,0.00435219,0.0007352806],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005015195,"about_ca_system_score_gemma":0.01042746,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002004359,"about_ca_topic_score_gemma":0.003608128,"domain_scores_codex":[0.9834599,0.009704266,0.001139834,0.001234297,0.003509631,0.0009521666],"domain_scores_gemma":[0.9658895,0.015916,0.00351245,0.003839129,0.006361803,0.004481236],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001879134,0.001322289,0.1176043,0.0004200417,0.00001422459,0.001776122,0.5055659,0.001008339,0.01141371,0.1326165,0.003184174,0.2248866],"study_design_scores_gemma":[0.0001099512,0.001013872,0.1571224,0.0008146871,0.000009832429,0.004280857,0.496395,0.007003862,0.007870055,0.1956552,0.1295457,0.0001786889],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8748407,0.0005194712,0.06165882,0.01209706,0.00007605874,0.0004014676,0.00008001733,0.0003785023,0.04994788],"genre_scores_gemma":[0.9613699,0.0001060921,0.03540342,0.0004674648,0.00001214106,0.0002296816,0.0000374544,0.00003789091,0.002335979],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01738891,"threshold_uncertainty_score":0.09196246,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2072296938","doi":"10.1080/10627190802394222","title":"Validity Issues in Assessing English Language Learners' Language Proficiency","year":2008,"lang":"en","type":"article","venue":"Educational Assessment","topic":"Educational Assessment and Pedagogy","field":"Social Sciences","cited_by":47,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Memorial University of Newfoundland","keywords":"Accountability; Legislation; Language proficiency; Language assessment; Psychology; English language; Pedagogy; Medical education; Mathematics education; Political science; Medicine","authors":[{"name":"Mikyung Kim Wolf","is_ca":false},{"name":"Tim Farnsworth","is_ca":false},{"name":"Joan L. Herman","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06769468706590388,"gpt":0.4642791324549266,"spread":0.3965844453890227,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.4755363,0.0007453884,0.001628255,0.007086383,0.006843224,0.008754864,0.003277047,0.002809954,0.001533514],"category_scores_gemma":[0.6634406,0.0008835891,0.00195967,0.006058418,0.02186871,0.009952812,0.008223538,0.004860862,0.0005631007],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00895578,"about_ca_system_score_gemma":0.02553042,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01127853,"about_ca_topic_score_gemma":0.01190938,"domain_scores_codex":[0.4212688,0.421712,0.04943635,0.01275552,0.09128568,0.003541731],"domain_scores_gemma":[0.239348,0.6323883,0.02509948,0.02553212,0.07563836,0.00199378],"domain_codex":"methods","domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004294566,0.0004156962,0.2184798,0.005036267,0.000792253,0.0004807914,0.05727706,0.002346057,0.001710829,0.3053556,0.01209895,0.3955773],"study_design_scores_gemma":[0.0003479783,0.00174166,0.210841,0.02288405,0.0009086266,0.001807904,0.06677267,0.02033404,0.01107296,0.5300572,0.1327744,0.0004575288],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2105878,0.02595357,0.4274227,0.1304906,0.005109836,0.008753577,0.0009393753,0.0004330454,0.1903095],"genre_scores_gemma":[0.7472276,0.004381286,0.2241248,0.01183519,0.001261517,0.007369042,0.0004529495,0.0002416676,0.003105988],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.4755363,"threshold_uncertainty_score":0.6467571,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3113373418","doi":"10.1080/10627197.2020.1858783","title":"The Effect of Linguistic Factors on Assessment of English Language Learners’ Mathematical Ability: A Differential Item Functioning Analysis","year":2020,"lang":"en","type":"article","venue":"Educational Assessment","topic":"Cognitive and developmental aspects of mathematical skills","field":"Mathematics","cited_by":25,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Differential item functioning; Ell; Psychology; Achievement test; Test (biology); Standardized test; Mathematics education; Language proficiency; Item response theory; Confirmatory factor analysis; Item analysis; Language assessment; Linguistics; Psychometrics; Developmental psychology; Teaching method; Mathematics; Statistics; Vocabulary development","authors":[{"name":"Stephanie Buono","is_ca":true},{"name":"Eunice Eunhee Jang","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02162447829000549,"gpt":0.3490916083046128,"spread":0.3274671300146073,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02356987,0.0007028677,0.0006190172,0.002833223,0.0007002875,0.001290583,0.0006114921,0.000421159,0.00143492],"category_scores_gemma":[0.07537299,0.0002847291,0.001581261,0.001285158,0.001522886,0.001069707,0.00174217,0.0008416782,0.0002817546],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006343159,"about_ca_system_score_gemma":0.0005814756,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002279338,"about_ca_topic_score_gemma":0.003294216,"domain_scores_codex":[0.9840047,0.00960501,0.001725758,0.001058342,0.003040787,0.0005652832],"domain_scores_gemma":[0.878773,0.101619,0.00614432,0.005686603,0.006244899,0.001532093],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0004339791,0.00007290818,0.9800463,0.00002476775,0.0002340141,0.00005238934,0.001124617,0.0001930273,0.001631616,0.0001191301,0.00008754099,0.01597979],"study_design_scores_gemma":[0.00001868115,0.0005629243,0.9944436,0.000021503,0.00008553814,0.0001871988,0.0008149584,0.001557357,0.001789329,0.0001957713,0.0003032111,0.00001985733],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.998004,0.00009972037,0.001133036,0.00004366739,0.000006739459,0.00001900121,0.00004417812,0.000008614525,0.00064104],"genre_scores_gemma":[0.998557,0.00002888115,0.001162578,0.00001769986,0.000004284322,0.00003001513,0.000066318,0.000005332454,0.000127787],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02356987,"threshold_uncertainty_score":0.1246508,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2086203028","doi":"10.1080/10627190802602384","title":"Identifying Potential Test Item Misalignment Using Student Verbal Reports","year":2008,"lang":"en","type":"article","venue":"Educational Assessment","topic":"Educational Strategies and Epistemologies","field":"Psychology","cited_by":11,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Centre for Advancing Health Outcomes; University of Alberta","funders":"","keywords":"Ambiguity; Psychology; Cognition; Construct (python library); Test (biology); Item analysis; Confidence interval; Cognitive psychology; Applied psychology; Psychometrics; Statistics; Developmental psychology; Computer science; Mathematics","authors":[{"name":"Jacqueline P. Leighton","is_ca":true},{"name":"Rebecca Gokiert","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08693562939927937,"gpt":0.4388979660022925,"spread":0.3519623366030131,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05404278,0.0009933667,0.001154457,0.004345702,0.0004572688,0.003007892,0.001043239,0.0007894047,0.00100204],"category_scores_gemma":[0.3248609,0.0004942814,0.0006451857,0.003979544,0.0009466823,0.002374246,0.002074054,0.001084017,0.0003525895],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006401995,"about_ca_system_score_gemma":0.0008559712,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004899047,"about_ca_topic_score_gemma":0.0005945046,"domain_scores_codex":[0.9312668,0.03541249,0.01432599,0.002643793,0.01551628,0.0008344899],"domain_scores_gemma":[0.5858323,0.2939548,0.07054771,0.02123011,0.02727257,0.001162469],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001338268,0.0004583326,0.7828666,0.0006248938,0.0004159256,0.0006908351,0.01089888,0.002349365,0.007441176,0.001720978,0.0007234465,0.1904713],"study_design_scores_gemma":[0.0001534656,0.004153951,0.8740264,0.0009243317,0.0004846578,0.003218679,0.01457736,0.046421,0.03994861,0.009946895,0.005823719,0.00032084],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9472997,0.0003707129,0.04944824,0.0001477177,0.00007271369,0.000431522,0.0002037751,0.0001930848,0.001832506],"genre_scores_gemma":[0.9745072,0.0001553443,0.02415849,0.00006773815,0.00003572717,0.0004420841,0.0003155747,0.00003492233,0.0002829187],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.05404278,"threshold_uncertainty_score":0.2858089,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}