{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":4,"total_is_capped":false,"direct_labels_cover":0,"predictions_cover":4,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"3202f460c37a","filters":{"venue":"Large-scale Assessments in Education"}},"results":[{"id":"W2589097440","doi":"10.1186/s40536-017-0042-x","title":"A structural equation modeling approach for examining position effects in large-scale assessments","year":2017,"lang":"en","type":"article","venue":"Large-scale Assessments in Education","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":17,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Structural equation modeling; Position (finance); Equivalence (formal languages); Item response theory; Scale (ratio); Multilevel model; Context (archaeology); Computer science; Econometrics; Position paper; Psychometrics; Statistics; Mathematics; Machine learning","authors":[{"name":"Okan Bulut","is_ca":true},{"name":"Qi Quo","is_ca":true},{"name":"Mark J. Gierl","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.344505919788792,"gpt":0.5186875194954802,"spread":0.1741815997066882,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02338569,0.001964945,0.001523816,0.004277646,0.001358816,0.002112125,0.002265544,0.001325249,0.005141938],"category_scores_gemma":[0.05969676,0.001076876,0.00282273,0.007990905,0.001222457,0.002556473,0.002221091,0.004276478,0.0007439648],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002325388,"about_ca_system_score_gemma":0.005089462,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009513557,"about_ca_topic_score_gemma":0.01411949,"domain_scores_codex":[0.9786819,0.016425,0.0008423775,0.001513351,0.00231534,0.00022207],"domain_scores_gemma":[0.9569303,0.03562627,0.00286317,0.00195863,0.002413615,0.0002080201],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002243848,0.001876126,0.1269922,0.001817959,0.004375704,0.0007694429,0.008571294,0.1338541,0.004002972,0.3429181,0.01094161,0.3636561],"study_design_scores_gemma":[0.0002855552,0.001993556,0.04950566,0.001022146,0.001289858,0.0004638625,0.003886443,0.609544,0.002491815,0.303539,0.02568608,0.0002920781],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02067523,0.0002241068,0.9736574,0.0005585609,0.00009635369,0.001285489,0.0007468443,0.0003030471,0.002452963],"genre_scores_gemma":[0.1704668,0.0005693887,0.8188707,0.0002214867,0.00007720151,0.007205861,0.001304969,0.0000841564,0.001199369],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02338569,"threshold_uncertainty_score":0.1236768,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4384942374","doi":"10.1186/s40536-023-00177-5","title":"Incorporating test-taking engagement into the item selection algorithm in low-stakes computerized adaptive tests","year":2023,"lang":"en","type":"article","venue":"Large-scale Assessments in Education","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":17,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Computerized adaptive testing; Test (biology); Selection (genetic algorithm); Psychology; Computer science; Trait; Machine learning; Psychometrics; Developmental psychology","authors":[{"name":"Guher Gorgun","is_ca":true},{"name":"Okan Bulut","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2262113854356192,"gpt":0.4864308640796258,"spread":0.2602194786440065,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03964259,0.001035428,0.001418287,0.003099415,0.0005250961,0.001521105,0.001678885,0.001125457,0.001857446],"category_scores_gemma":[0.1131489,0.0005403416,0.0008575427,0.002082913,0.000981523,0.001810733,0.001349593,0.001774012,0.0004709092],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009587625,"about_ca_system_score_gemma":0.001662758,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001755182,"about_ca_topic_score_gemma":0.003218368,"domain_scores_codex":[0.9770939,0.01822634,0.001024304,0.001243223,0.002126922,0.0002853375],"domain_scores_gemma":[0.8586547,0.1240624,0.00521726,0.003905155,0.007043929,0.001116461],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001533262,0.00128397,0.1409973,0.0004133955,0.000641449,0.0001023556,0.0008911873,0.08301967,0.004839596,0.003648366,0.001441276,0.7611881],"study_design_scores_gemma":[0.0002591106,0.001281035,0.03628886,0.0001664674,0.000160558,0.00008154375,0.0002111047,0.9466958,0.005813603,0.007769854,0.001178564,0.00009352434],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2642927,0.0003310356,0.7316007,0.0003585262,0.00003710019,0.001061494,0.0001137742,0.001059289,0.001145408],"genre_scores_gemma":[0.6008786,0.00007787329,0.3976533,0.0001054348,0.00001785786,0.0007349714,0.0001807098,0.00004276593,0.0003084618],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03964259,"threshold_uncertainty_score":0.2096525,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3011227050","doi":"10.1186/s40536-020-00080-3","title":"Are large surveys of adult literacy skills as comparable over time as we think?","year":2020,"lang":"en","type":"article","venue":"Large-scale Assessments in Education","topic":"Education Systems and Policy","field":"Social Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true},"ca_institutions":"Institut National de la Recherche Scientifique","funders":"Canadian Institutes of Health Research; Social Sciences and Humanities Research Council of Canada; Fonds de Recherche du Québec-Société et Culture","keywords":"Ceteris paribus; Literacy; Cohort; Adult literacy; Cohort effect; Survey data collection; Psychology; Demographic economics; Demography; Medicine; Pedagogy; Sociology; Economics; Statistics","authors":[{"name":"Samuel Vézina","is_ca":true},{"name":"Alain Bélanger","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02048207184716733,"gpt":0.4094220301978794,"spread":0.388939958350712,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1345872,0.0005874601,0.001641483,0.003480117,0.001354173,0.006354971,0.003059705,0.002354123,0.003414818],"category_scores_gemma":[0.4719467,0.0006588154,0.001382306,0.007134188,0.005217103,0.006132192,0.002972542,0.002347586,0.0007944495],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003754258,"about_ca_system_score_gemma":0.005754689,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.1227949,"about_ca_topic_score_gemma":0.07014492,"domain_scores_codex":[0.8754196,0.08143558,0.007281539,0.01130082,0.02204553,0.002516797],"domain_scores_gemma":[0.5529637,0.2062105,0.08692878,0.08253399,0.06647388,0.004889194],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0009030809,0.0001548521,0.7359579,0.002340063,0.004821843,0.0002363295,0.01061797,0.002851372,0.0008175873,0.04327318,0.0520443,0.1459816],"study_design_scores_gemma":[0.0002182718,0.0004109711,0.8781281,0.00485424,0.001337891,0.0001729398,0.007866797,0.00334662,0.001889626,0.02841744,0.07315791,0.0001992804],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.569065,0.05034678,0.0921889,0.1972982,0.01120663,0.001192943,0.02177083,0.0008175336,0.05611314],"genre_scores_gemma":[0.9551619,0.003763508,0.01619509,0.01724384,0.001600623,0.0005897747,0.003329048,0.000144868,0.001971293],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8654128,"threshold_uncertainty_score":0.7117738,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4392361000","doi":"10.1186/s40536-024-00194-y","title":"An engagement-aware predictive model to evaluate problem-solving performance from the study of adult skills' (PIAAC 2012) process data","year":2024,"lang":"en","type":"article","venue":"Large-scale Assessments in Education","topic":"Online Learning and Analytics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Task (project management); Scale (ratio); Process (computing); Learning analytics; Computer science; Test (biology); Analytics; Psychology; Artificial intelligence; Data science","authors":[{"name":"Jinnie Shin","is_ca":false},{"name":"Bowen Wang","is_ca":false},{"name":"Wallace Nascimento Pinto","is_ca":false},{"name":"Mark J. Gierl","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02761815327562231,"gpt":0.3969492216172661,"spread":0.3693310683416438,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002938673,0.0006598384,0.0004864766,0.001320849,0.000273508,0.001104638,0.0007769424,0.0006946602,0.001225387],"category_scores_gemma":[0.0127836,0.0002710497,0.0005255385,0.0007698631,0.0003171875,0.000917747,0.001003882,0.001221411,0.0002710675],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000898128,"about_ca_system_score_gemma":0.0007671429,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009977049,"about_ca_topic_score_gemma":0.0107866,"domain_scores_codex":[0.9991871,0.0003724235,0.00004783318,0.0002146734,0.0001180486,0.00005986234],"domain_scores_gemma":[0.9943299,0.004075677,0.0004890602,0.000367885,0.0005362416,0.0002012273],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005379864,0.001231124,0.1983843,0.0001145433,0.0002602345,0.000171971,0.0005695572,0.6491005,0.002676909,0.002800175,0.001581001,0.1425717],"study_design_scores_gemma":[0.000004397978,0.00005032614,0.009003429,0.000007812045,0.000008432805,0.000008440754,0.00002477152,0.9893956,0.0003175703,0.001063332,0.000109538,0.000006305564],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7658622,0.000162197,0.2300134,0.000456516,0.00002570327,0.0001992301,0.001048147,0.0006433053,0.001589254],"genre_scores_gemma":[0.9794862,0.00003468672,0.01929,0.00002806917,0.000006543674,0.0001043748,0.0005732208,0.00000945408,0.0004675661],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009977049,"threshold_uncertainty_score":0.01983792,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}