{"meta":{"query_hash":"3202f460c37a","filters":{"venue":"Large-scale Assessments in Education"},"cohort_total":4,"direct_labels_cover":0,"predictions_cover":4,"exported":4,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/3202f460c37a","api":"https://metacan.xera.ac/api/v1/cohort?venue=Large-scale+Assessments+in+Education"},"results":[{"id":"W2589097440","doi":"10.1186/s40536-017-0042-x","title":"A structural equation modeling approach for examining position effects in large-scale assessments","year":2017,"lang":"en","type":"article","venue":"Large-scale Assessments in Education","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Structural equation modeling; Position (finance); Equivalence (formal languages); Item response theory; Scale (ratio); Multilevel model; Context (archaeology); Computer science; Econometrics; Position paper; Psychometrics; Statistics; Mathematics; Machine learning","score_opus":0.34450591978879197,"score_gpt":0.5186875194954802,"score_spread":0.17418159970668823,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2589097440","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020675234,0.00022410683,0.9736574,0.00055856095,0.000096353695,0.0012854893,0.00074684434,0.00030304713,0.002452963],"genre_scores_gemma":[0.17046683,0.0005693887,0.81887066,0.00022148671,0.00007720151,0.007205861,0.0013049691,0.000084156396,0.0011993687],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.97868186,0.016425002,0.00084237754,0.0015133506,0.00231534,0.00022206997],"domain_scores_gemma":[0.9569303,0.03562627,0.0028631703,0.0019586296,0.0024136147,0.00020802005],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.023385692,0.0019649446,0.001523816,0.004277646,0.0013588158,0.0021121246,0.0022655441,0.001325249,0.0051419376],"category_scores_gemma":[0.05969676,0.0010768756,0.0028227295,0.007990905,0.0012224574,0.0025564728,0.0022210914,0.004276478,0.0007439648],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022438484,0.0018761256,0.12699217,0.0018179585,0.0043757036,0.0007694429,0.008571294,0.13385406,0.004002972,0.34291813,0.010941607,0.3636561],"study_design_scores_gemma":[0.00028555523,0.0019935556,0.049505662,0.001022146,0.0012898581,0.0004638625,0.0038864426,0.609544,0.0024918152,0.30353904,0.025686083,0.00029207813],"about_ca_topic_score_codex":0.009513557,"about_ca_topic_score_gemma":0.014119486,"teacher_disagreement_score":0.023385692,"about_ca_system_score_codex":0.0023253884,"about_ca_system_score_gemma":0.005089462,"threshold_uncertainty_score":0.12367684},"labels":[],"label_agreement":null},{"id":"W3011227050","doi":"10.1186/s40536-020-00080-3","title":"Are large surveys of adult literacy skills as comparable over time as we think?","year":2020,"lang":"en","type":"article","venue":"Large-scale Assessments in Education","topic":"Education Systems and Policy","field":"Social Sciences","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Institut National de la Recherche Scientifique","funders":"Canadian Institutes of Health Research; Social Sciences and Humanities Research Council of Canada; Fonds de Recherche du Québec-Société et Culture","keywords":"Ceteris paribus; Literacy; Cohort; Adult literacy; Cohort effect; Survey data collection; Psychology; Demographic economics; Demography; Medicine; Pedagogy; Sociology; Economics; Statistics","score_opus":0.020482071847167334,"score_gpt":0.40942203019787937,"score_spread":0.38893995835071205,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3011227050","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.569065,0.050346784,0.0921889,0.19729824,0.011206632,0.0011929431,0.021770833,0.0008175336,0.056113135],"genre_scores_gemma":[0.95516187,0.0037635083,0.016195087,0.01724384,0.0016006225,0.0005897747,0.0033290477,0.00014486804,0.0019712925],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.8754196,0.081435576,0.0072815386,0.011300817,0.02204553,0.0025167973],"domain_scores_gemma":[0.55296373,0.20621046,0.086928785,0.082533985,0.06647388,0.0048891944],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.13458723,0.0005874601,0.0016414828,0.003480117,0.0013541727,0.0063549713,0.0030597053,0.0023541232,0.0034148183],"category_scores_gemma":[0.47194675,0.00065881544,0.0013823056,0.007134188,0.0052171033,0.0061321924,0.0029725416,0.0023475864,0.0007944495],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009030809,0.00015485212,0.73595786,0.0023400625,0.004821843,0.00023632946,0.010617973,0.0028513724,0.0008175873,0.043273184,0.052044302,0.14598161],"study_design_scores_gemma":[0.00021827179,0.00041097114,0.87812805,0.0048542395,0.0013378906,0.00017293978,0.007866797,0.0033466201,0.0018896256,0.028417436,0.073157914,0.00019928039],"about_ca_topic_score_codex":0.12279492,"about_ca_topic_score_gemma":0.07014492,"teacher_disagreement_score":0.8654128,"about_ca_system_score_codex":0.003754258,"about_ca_system_score_gemma":0.005754689,"threshold_uncertainty_score":0.7117738},"labels":[],"label_agreement":null},{"id":"W4384942374","doi":"10.1186/s40536-023-00177-5","title":"Incorporating test-taking engagement into the item selection algorithm in low-stakes computerized adaptive tests","year":2023,"lang":"en","type":"article","venue":"Large-scale Assessments in Education","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computerized adaptive testing; Test (biology); Selection (genetic algorithm); Psychology; Computer science; Trait; Machine learning; Psychometrics; Developmental psychology","score_opus":0.22621138543561922,"score_gpt":0.48643086407962577,"score_spread":0.2602194786440065,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4384942374","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26429266,0.00033103558,0.7316007,0.00035852622,0.000037100188,0.0010614942,0.00011377423,0.0010592891,0.0011454078],"genre_scores_gemma":[0.6008786,0.00007787329,0.39765334,0.00010543483,0.000017857863,0.0007349714,0.00018070979,0.00004276593,0.00030846178],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9770939,0.018226344,0.0010243043,0.0012432225,0.0021269224,0.0002853375],"domain_scores_gemma":[0.8586547,0.12406241,0.0052172597,0.0039051548,0.007043929,0.0011164614],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.039642587,0.0010354276,0.0014182869,0.0030994155,0.00052509614,0.0015211051,0.001678885,0.0011254568,0.0018574463],"category_scores_gemma":[0.113148876,0.0005403416,0.00085754273,0.0020829132,0.000981523,0.0018107332,0.001349593,0.001774012,0.00047090923],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015332615,0.0012839702,0.14099732,0.00041339552,0.000641449,0.00010235561,0.00089118734,0.08301967,0.004839596,0.0036483663,0.0014412763,0.76118815],"study_design_scores_gemma":[0.00025911062,0.0012810349,0.036288857,0.00016646745,0.00016055805,0.000081543745,0.00021110474,0.9466958,0.005813603,0.0077698543,0.0011785644,0.000093524344],"about_ca_topic_score_codex":0.0017551817,"about_ca_topic_score_gemma":0.0032183682,"teacher_disagreement_score":0.039642587,"about_ca_system_score_codex":0.00095876254,"about_ca_system_score_gemma":0.001662758,"threshold_uncertainty_score":0.20965254},"labels":[],"label_agreement":null},{"id":"W4392361000","doi":"10.1186/s40536-024-00194-y","title":"An engagement-aware predictive model to evaluate problem-solving performance from the study of adult skills' (PIAAC 2012) process data","year":2024,"lang":"en","type":"article","venue":"Large-scale Assessments in Education","topic":"Online Learning and Analytics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Task (project management); Scale (ratio); Process (computing); Learning analytics; Computer science; Test (biology); Analytics; Psychology; Artificial intelligence; Data science","score_opus":0.02761815327562231,"score_gpt":0.3969492216172661,"score_spread":0.36933106834164375,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392361000","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7658622,0.00016219703,0.23001336,0.000456516,0.000025703268,0.00019923005,0.0010481474,0.0006433053,0.0015892537],"genre_scores_gemma":[0.9794862,0.000034686724,0.019290002,0.000028069167,0.0000065436743,0.00010437479,0.0005732208,0.00000945408,0.00046756613],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99918705,0.00037242347,0.00004783318,0.00021467336,0.00011804858,0.00005986234],"domain_scores_gemma":[0.9943299,0.0040756767,0.0004890602,0.00036788502,0.00053624157,0.00020122729],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029386727,0.00065983844,0.00048647658,0.0013208488,0.00027350802,0.0011046381,0.0007769424,0.0006946602,0.0012253871],"category_scores_gemma":[0.012783602,0.00027104968,0.0005255385,0.00076986314,0.00031718746,0.00091774703,0.0010038816,0.001221411,0.00027106752],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005379864,0.0012311238,0.19838427,0.00011454326,0.00026023446,0.00017197095,0.0005695572,0.64910054,0.0026769089,0.0028001748,0.0015810007,0.14257166],"study_design_scores_gemma":[0.0000043979776,0.000050326136,0.009003429,0.000007812045,0.000008432805,0.000008440754,0.000024771522,0.98939556,0.00031757026,0.001063332,0.000109538014,0.0000063055636],"about_ca_topic_score_codex":0.009977049,"about_ca_topic_score_gemma":0.010786598,"teacher_disagreement_score":0.009977049,"about_ca_system_score_codex":0.000898128,"about_ca_system_score_gemma":0.0007671429,"threshold_uncertainty_score":0.019837916},"labels":[],"label_agreement":null}]}