{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":3,"total_is_capped":false,"direct_labels_cover":0,"predictions_cover":3,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"9bc749b77fef","filters":{"venue":"Studies in Language Assessment"}},"results":[{"id":"W4376626995","doi":"10.58379/rshg8366","title":"DIF investigations across groups of gender and academic background in a large-scale high-stakes language test ","year":2015,"lang":"en","type":"article","venue":"Studies in Language Assessment","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"","funders":"Queen's University","keywords":"Test (biology); Differential item functioning; Psychology; Quality (philosophy); Reliability (semiconductor); Scale (ratio); Gender bias; Social psychology; Applied psychology; Mathematics education; Medical education; Item response theory; Developmental psychology; Psychometrics; Medicine","authors":[{"name":"Xiamei Song","is_ca":false},{"name":"Liying Cheng","is_ca":false},{"name":"Don A. Klinger","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.119758425185111,"gpt":0.46502313377722,"spread":0.3452647085921091,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04040404,0.0004531023,0.0006980777,0.003450269,0.001270335,0.001071426,0.0007638202,0.0004919579,0.001464676],"category_scores_gemma":[0.1081698,0.0001985538,0.0009474493,0.001651765,0.00162525,0.00119332,0.002145661,0.0005769636,0.0002655243],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00105017,"about_ca_system_score_gemma":0.001098234,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002136123,"about_ca_topic_score_gemma":0.00292848,"domain_scores_codex":[0.9715961,0.01478039,0.003445428,0.001990824,0.006804809,0.001382522],"domain_scores_gemma":[0.9158825,0.04578331,0.01125835,0.007485604,0.01783516,0.001755119],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0004037235,0.0002618251,0.943424,0.00008044475,0.0001515599,0.0001201539,0.01263108,0.0001380861,0.001441122,0.0007012446,0.0002735658,0.04037327],"study_design_scores_gemma":[0.00003085711,0.0007357008,0.982946,0.00006945791,0.00007142502,0.0002558158,0.01026507,0.001386553,0.002000309,0.001024997,0.001177388,0.0000363734],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9965592,0.00008832804,0.002013211,0.00009053443,0.00002433701,0.0001003533,0.00005718681,0.000006016672,0.001060787],"genre_scores_gemma":[0.9984264,0.00002846117,0.001099587,0.00005161251,0.00001413347,0.00008854158,0.00008057636,0.000004211841,0.000206378],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04040404,"threshold_uncertainty_score":0.2136796,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4376626656","doi":"10.58379/dayb9070","title":"Development of a Spanish generic writing skills scale for the Colombian Graduate Skills Assessment (Saber Pro)","year":2015,"lang":"en","type":"article","venue":"Studies in Language Assessment","topic":"Writing and Handwriting Education","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"La Trobe University","keywords":"Graduation (instrument); Scale (ratio); Context (archaeology); Mathematics education; Scripting language; Psychology; Trait; Test (biology); Writing assessment; Reliability (semiconductor); Pedagogy; Computer science; Geography; Mathematics; Cartography","authors":[{"name":"Ana María Ducasse","is_ca":false},{"name":"Kathryn Hill","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1354379703205154,"gpt":0.4766824069376233,"spread":0.3412444366171078,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003223638,0.0004946049,0.0003790556,0.002015422,0.000412768,0.0008000348,0.0005660232,0.0002979068,0.003517418],"category_scores_gemma":[0.009926713,0.000190432,0.0003641264,0.000680071,0.0003277992,0.0004957989,0.0009171916,0.0006890584,0.001084268],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009099208,"about_ca_system_score_gemma":0.002108906,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006939907,"about_ca_topic_score_gemma":0.01999586,"domain_scores_codex":[0.9983077,0.0005541006,0.0002556186,0.0001498777,0.0006426754,0.00008998938],"domain_scores_gemma":[0.9967429,0.0005892074,0.0004240321,0.0001643584,0.001823624,0.0002558638],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002012321,0.0006831175,0.3266545,0.001019108,0.00008646634,0.0006712145,0.007797374,0.001333759,0.01247409,0.001774505,0.02555291,0.6217517],"study_design_scores_gemma":[0.0001043492,0.0005775915,0.9160292,0.0004745568,0.00004540657,0.00104401,0.006033101,0.004107222,0.002984886,0.001127011,0.06737123,0.0001013491],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8767866,0.001330596,0.04185243,0.001664494,0.0005238595,0.01114458,0.006960758,0.0009079588,0.05882862],"genre_scores_gemma":[0.76661,0.001273581,0.2017302,0.0003859824,0.00008919252,0.01255894,0.005011203,0.0001517948,0.01218913],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006939907,"threshold_uncertainty_score":0.01704848,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4367592383","doi":"10.58379/homq5772","title":"Exploring shared and individual assessment of paired oral interactions","year":2022,"lang":"en","type":"article","venue":"Studies in Language Assessment","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":1,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Concordia University","funders":"","keywords":"Rubric; Psychology; Task (project management); Variation (astronomy); Modality (human–computer interaction); Audiology; Cognitive psychology; Computer science; Medicine; Artificial intelligence; Mathematics education","authors":[{"name":"Pakize Uludag","is_ca":true},{"name":"Kim McDonough","is_ca":true},{"name":"Pavel Trofimovich","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3685810683728511,"gpt":0.4261905241132528,"spread":0.05760945574040166,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00932718,0.0005056211,0.0006032001,0.002414881,0.0006061844,0.002478557,0.0007519381,0.0004676426,0.002192436],"category_scores_gemma":[0.04092171,0.0002431327,0.0004022984,0.0007571854,0.001005651,0.001187299,0.003826073,0.0005798872,0.0004170052],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004830698,"about_ca_system_score_gemma":0.0005547241,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001212655,"about_ca_topic_score_gemma":0.003078242,"domain_scores_codex":[0.9892836,0.005354604,0.0009180986,0.001248075,0.002735716,0.0004600575],"domain_scores_gemma":[0.962162,0.02170787,0.006308435,0.002396072,0.005992746,0.001432903],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"qualitative","study_design_scores_codex":[0.00139071,0.000556397,0.5449762,0.0007459543,0.0007496392,0.0009478261,0.1027713,0.001579835,0.05835819,0.0007985043,0.0007005083,0.286425],"study_design_scores_gemma":[0.00003710448,0.001897952,0.9235268,0.0001422021,0.000166378,0.001653077,0.04202495,0.00529615,0.02099677,0.001242391,0.002841497,0.0001747842],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9906842,0.0001302678,0.005811508,0.00002517368,0.00001252869,0.00007178082,0.00007255848,0.00003587992,0.003156231],"genre_scores_gemma":[0.9950265,0.00007098489,0.004021083,0.00001143434,0.00000907299,0.00008425694,0.0000775342,0.00001538376,0.0006837272],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.00932718,"threshold_uncertainty_score":0.04932737,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}