{"meta":{"query_hash":"8eedc4ff3541","filters":{"venue":"Language Assessment Quarterly"},"cohort_total":22,"direct_labels_cover":0,"predictions_cover":22,"exported":22,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/8eedc4ff3541","api":"https://metacan.xera.ac/api/v1/cohort?venue=Language+Assessment+Quarterly"},"results":[{"id":"W2002561419","doi":"10.1080/15434303.2014.936603","title":"Using Lexical Profiling Tools to Investigate Children’s Written Vocabulary in Grade 3: An Exploratory Study","year":2015,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Writing and Handwriting Education","field":"Social Sciences","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; University of Calgary","funders":"","keywords":"Vocabulary; Rubric; Lexical diversity; Salient; Psychology; Computer science; Linguistics; Profiling (computer programming); Exploratory research; Vocabulary development; Natural language processing; Lexical density; Trait; Artificial intelligence; Mathematics education; Lexical item; Sociology","score_opus":0.12772409961458991,"score_gpt":0.43022945482507097,"score_spread":0.30250535521048105,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2002561419","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99951816,0.000023868555,0.00007612131,0.000005214041,7.0296096e-7,0.00002028983,0.000044817356,0.0000033376955,0.0003075753],"genre_scores_gemma":[0.9985531,0.00007456695,0.00064642733,0.000016412587,0.0000015436816,0.000093053444,0.00011543826,0.000005291793,0.00049420615],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9982528,0.0003965021,0.00026485402,0.00027177428,0.00045998555,0.00035406926],"domain_scores_gemma":[0.99527895,0.0015184288,0.0011866273,0.0003573123,0.0011791167,0.00047961067],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028503148,0.0007036569,0.00089181087,0.0028111076,0.0012720648,0.0023690194,0.00078154774,0.00077176595,0.00093427737],"category_scores_gemma":[0.0073390394,0.0004934218,0.0006881803,0.001456923,0.0009905923,0.0011208876,0.0017435057,0.0009924689,0.0005055233],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001436547,0.0008237435,0.83256984,0.00012708035,0.00003999272,0.001497062,0.12941964,0.00011124684,0.008976339,0.00016392652,0.00017611319,0.025951393],"study_design_scores_gemma":[0.000008310734,0.00094865775,0.9523659,0.0000432082,0.00002849524,0.00083807344,0.042230397,0.00014401496,0.0020256084,0.000087878645,0.0012570814,0.00002238704],"about_ca_topic_score_codex":0.013313918,"about_ca_topic_score_gemma":0.027632885,"teacher_disagreement_score":0.013313918,"about_ca_system_score_codex":0.0008771924,"about_ca_system_score_gemma":0.0010792268,"threshold_uncertainty_score":0.026472867},"labels":[],"label_agreement":null},{"id":"W2015223621","doi":"10.1080/15434303.2014.981334","title":"Interpreting the Impact of the Ontario Secondary School Literacy Test on Second Language Students Within an Argument-Based Validation Framework","year":2015,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Queen's University","funders":"","keywords":"Argument (complex analysis); Literacy; Context (archaeology); Mathematics education; Test (biology); Curriculum; Reading (process); Ell; Language proficiency; Empirical research; Psychology; Pedagogy; Language assessment; Linguistics; Teaching method; Mathematics","score_opus":0.016460365348475472,"score_gpt":0.3890332495887422,"score_spread":0.37257288424026674,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2015223621","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.87345415,0.0069315084,0.014462142,0.033461295,0.0005854814,0.0012062998,0.0015701172,0.00006345061,0.06826553],"genre_scores_gemma":[0.992042,0.0004945359,0.004812666,0.00092674297,0.00004284059,0.0005099642,0.0002940349,0.00001919496,0.0008579599],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.8053914,0.13100895,0.0076246806,0.004572826,0.04813973,0.0032623992],"domain_scores_gemma":[0.35160136,0.5002401,0.06280159,0.012803408,0.07066703,0.001886627],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.17836873,0.0008130009,0.0010102672,0.0067225182,0.00523628,0.009038216,0.0048267064,0.002168444,0.0024857516],"category_scores_gemma":[0.5317957,0.0006734873,0.0016412156,0.0061114575,0.01809934,0.0041334443,0.008799448,0.002775795,0.000191719],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010975422,0.00033355367,0.64067364,0.0028158524,0.0009401556,0.0008947505,0.1903153,0.002683178,0.00074478285,0.0645788,0.006554093,0.08836835],"study_design_scores_gemma":[0.00036641525,0.00092820055,0.7094768,0.011494198,0.0013893165,0.00019027981,0.19120882,0.01114879,0.004854373,0.030235814,0.0384125,0.00029437154],"about_ca_topic_score_codex":0.45774013,"about_ca_topic_score_gemma":0.44409764,"teacher_disagreement_score":0.5422599,"about_ca_system_score_codex":0.045906212,"about_ca_system_score_gemma":0.04270188,"threshold_uncertainty_score":0.94331527},"labels":[],"label_agreement":null},{"id":"W2028824355","doi":"10.1080/15434300801934751","title":"Comments on “Evaluation of the Usefulness of the<i>Versant for English</i>Test: A Response”: The Author Responds","year":2008,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Institute for Christian Studies","funders":"","keywords":"Test (biology); Psychology; English language; Language assessment; Mathematics education; Linguistics; Computer science; Cognitive psychology","score_opus":0.0786983967731293,"score_gpt":0.32736721141316966,"score_spread":0.24866881464004037,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2028824355","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.005004733,0.00057474565,0.0013359503,0.93617487,0.043371182,0.00025959854,0.0006235459,0.0006279194,0.012027527],"genre_scores_gemma":[0.018849535,0.00049925386,0.0013370401,0.92940253,0.008572477,0.00032517928,0.0002694968,0.00028711243,0.040457405],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9934301,0.0017875625,0.00089692057,0.00056315033,0.0027037375,0.0006185037],"domain_scores_gemma":[0.9502862,0.024037825,0.0019857823,0.0006646313,0.020742591,0.0022830216],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007785209,0.0007931348,0.0007018366,0.0008130772,0.0037672084,0.0020682083,0.0020014362,0.017300151,0.0220489],"category_scores_gemma":[0.06828541,0.00051649404,0.0010257586,0.0005715948,0.0018031965,0.0017762132,0.0022297844,0.0113785295,0.010458796],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006607659,0.000022011638,0.00074695575,0.0000810778,0.000007650087,0.0002840449,0.00084096217,0.00007568295,0.0005888028,0.00044351147,0.99211746,0.004725642],"study_design_scores_gemma":[0.0000497866,0.0002185223,0.005076589,0.00037312703,0.000048275124,0.00053524517,0.006613724,0.0003863351,0.002398401,0.0008301411,0.983313,0.00015689859],"about_ca_topic_score_codex":0.014217019,"about_ca_topic_score_gemma":0.017446302,"teacher_disagreement_score":0.0220489,"about_ca_system_score_codex":0.00392913,"about_ca_system_score_gemma":0.004885445,"threshold_uncertainty_score":0.07376087},"labels":[],"label_agreement":null},{"id":"W2096160735","doi":"10.1080/15434300701375832","title":"Three Generations of DIF Analyses: Considering Where It Has Been, Where It Is Now, and Where It Is Going","year":2007,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":386,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Differential item functioning; Contrast (vision); Psychology; Cognitive psychology; Item response theory; Econometrics; Computer science; Psychometrics; Artificial intelligence; Mathematics; Developmental psychology","score_opus":0.08492024923436474,"score_gpt":0.43632217501312276,"score_spread":0.351401925778758,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2096160735","genre_codex":"methods","genre_gemma":"review","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14835806,0.06134102,0.6552822,0.057712846,0.0033942328,0.0021760806,0.0009970332,0.0011333197,0.06960519],"genre_scores_gemma":[0.601673,0.01622359,0.3663352,0.007305851,0.0014434084,0.0023526666,0.00075700483,0.00032766323,0.0035816485],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.917119,0.057676245,0.006607662,0.0040642447,0.013312986,0.0012199016],"domain_scores_gemma":[0.74169904,0.2046002,0.011955902,0.01647621,0.022233516,0.0030351824],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.10416991,0.0017919426,0.0019397083,0.019278677,0.004767225,0.0121383555,0.002440873,0.0021572895,0.0032664526],"category_scores_gemma":[0.22207719,0.0011627283,0.0015935525,0.0117942095,0.015375516,0.013985605,0.010856341,0.007929076,0.0006842125],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018384919,0.00012045936,0.082341604,0.0012719302,0.0002630008,0.0003471134,0.051327687,0.00067329523,0.0009597238,0.37195486,0.010003657,0.48055282],"study_design_scores_gemma":[0.000044229986,0.00035765098,0.054900095,0.0033290274,0.00024130325,0.002001452,0.052682813,0.006659271,0.001163268,0.7976819,0.080536135,0.0004028425],"about_ca_topic_score_codex":0.0050837114,"about_ca_topic_score_gemma":0.0061287005,"teacher_disagreement_score":0.8958301,"about_ca_system_score_codex":0.005226667,"about_ca_system_score_gemma":0.004919982,"threshold_uncertainty_score":0.55090976},"labels":[],"label_agreement":null},{"id":"W2112485328","doi":"10.1080/15434303.2015.1010726","title":"Teachers’ Grading Decision Making: Multiple Influencing Factors and Methods","year":2015,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":138,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Grading (engineering); Psychology; Mathematics education; English language; Multivariate analysis of variance; Statistics; Engineering; Mathematics","score_opus":0.04392888403571169,"score_gpt":0.44262896175885824,"score_spread":0.39870007772314653,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2112485328","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9835622,0.00042026726,0.009157084,0.0002548019,0.0000371652,0.000571558,0.00013652789,0.000074985175,0.0057854746],"genre_scores_gemma":[0.99441236,0.00008485719,0.0046064835,0.00002219572,0.000011508457,0.00013744262,0.00006755045,0.000014287144,0.00064328156],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.97926325,0.008851162,0.00219875,0.0022471333,0.0062029515,0.0012366921],"domain_scores_gemma":[0.9174977,0.052410282,0.015219703,0.0032044644,0.008862635,0.002805194],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016898522,0.0009483484,0.00083015655,0.0033329576,0.0017018814,0.0039618863,0.0009623157,0.00049513887,0.0028090216],"category_scores_gemma":[0.049344543,0.0005961421,0.0011446865,0.002893308,0.0011635238,0.0011623224,0.0013402426,0.00072215544,0.00025820525],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001997072,0.00021418264,0.9457048,0.00016802973,0.00021292889,0.0002285701,0.012216184,0.00047802617,0.0007314763,0.00048748698,0.00031785245,0.039040763],"study_design_scores_gemma":[0.00003721968,0.0002177043,0.97367686,0.00015583588,0.00034590953,0.00027309285,0.011679246,0.008557175,0.0013823548,0.0012701555,0.0023108406,0.000093682764],"about_ca_topic_score_codex":0.012489276,"about_ca_topic_score_gemma":0.015998064,"teacher_disagreement_score":0.016898522,"about_ca_system_score_codex":0.002369377,"about_ca_system_score_gemma":0.004415166,"threshold_uncertainty_score":0.089369},"labels":[],"label_agreement":null},{"id":"W2955380721","doi":"10.1080/15434303.2019.1628238","title":"“Be a Machine”: International Graduate Students’ Narratives around High-Stakes English Tests","year":2019,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Second Language Learning and Teaching","field":"Arts and Humanities","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Test of English as a Foreign Language; Narrative; Test (biology); Agency (philosophy); Language proficiency; Mathematics education; Language assessment; Psychology; Pedagogy; Study abroad; Medical education; Sociology; Linguistics; Social science","score_opus":0.021598236431953518,"score_gpt":0.29817958750423507,"score_spread":0.27658135107228154,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2955380721","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96600914,0.0010234931,0.0010293379,0.013112012,0.00028079306,0.00006699381,0.00013084493,0.000040259565,0.018307157],"genre_scores_gemma":[0.9912702,0.00084352016,0.0002723977,0.0016995902,0.00007235037,0.00004123701,0.00006957157,0.00007118148,0.005659981],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.98541707,0.008440397,0.00038958833,0.0008882053,0.0019441128,0.00292072],"domain_scores_gemma":[0.97894555,0.010250057,0.00219187,0.00072951324,0.0018427524,0.0060402774],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013705305,0.0018356738,0.0011810068,0.002439241,0.027127422,0.023819024,0.0038792107,0.0055300277,0.0032619794],"category_scores_gemma":[0.024818601,0.0011061745,0.000763237,0.0022554724,0.036671724,0.011241418,0.018085107,0.017496765,0.000847611],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000011870349,0.000012857793,0.001080786,0.000010365023,0.000001837125,0.0002114984,0.9960796,0.000010421032,0.00009896464,0.001221692,0.00058245513,0.0006776854],"study_design_scores_gemma":[0.0000011929761,0.000012782889,0.00047128505,0.000027694217,0.0000020878879,0.00007472221,0.9941195,0.00001663512,0.00006947792,0.00014380779,0.0050473777,0.000013374689],"about_ca_topic_score_codex":0.06362532,"about_ca_topic_score_gemma":0.09987503,"teacher_disagreement_score":0.06362532,"about_ca_system_score_codex":0.016150547,"about_ca_system_score_gemma":0.008569967,"threshold_uncertainty_score":0.12650996},"labels":[],"label_agreement":null},{"id":"W2975930148","doi":"10.1080/15434303.2019.1671392","title":"Incorporating Translanguaging in Language Assessment: The Case of a Test for University Professors","year":2019,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Second Language Learning and Teaching","field":"Arts and Humanities","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Translanguaging; Operationalization; Active listening; Test (biology); Psychology; Task (project management); Competence (human resources); Mathematics education; Pedagogy; Computer science; Social psychology","score_opus":0.012807469388652586,"score_gpt":0.28008995809552417,"score_spread":0.2672824887068716,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2975930148","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8727782,0.00077890255,0.08489715,0.007754609,0.00034165493,0.0012706314,0.000079271675,0.0010177385,0.031081837],"genre_scores_gemma":[0.8081688,0.00038236126,0.17904586,0.0014514257,0.00008832728,0.00039982342,0.00008731456,0.00032119625,0.010054906],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9325162,0.050799116,0.0024418316,0.0021320756,0.008638538,0.003472261],"domain_scores_gemma":[0.9257531,0.043732665,0.0028718752,0.00539998,0.015676249,0.0065661464],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.048535425,0.0010847042,0.00081832724,0.0018129599,0.006970806,0.0075758896,0.0034017041,0.003385118,0.001890314],"category_scores_gemma":[0.093260504,0.0007631566,0.0006507157,0.0014112311,0.005577439,0.0040975446,0.0071741166,0.004080228,0.0012213192],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00062233245,0.0017098311,0.04733107,0.0008749393,0.00006717374,0.016263854,0.402647,0.004127084,0.033202775,0.009034125,0.00516914,0.47895065],"study_design_scores_gemma":[0.00039496092,0.0073667546,0.064189926,0.0015357941,0.00026832495,0.027014116,0.36445892,0.026816597,0.11808568,0.015372083,0.3733196,0.0011772377],"about_ca_topic_score_codex":0.033682123,"about_ca_topic_score_gemma":0.09645731,"teacher_disagreement_score":0.048535425,"about_ca_system_score_codex":0.006768251,"about_ca_system_score_gemma":0.013738088,"threshold_uncertainty_score":0.25668293},"labels":[],"label_agreement":null},{"id":"W3111103222","doi":"10.1080/15434303.2020.1846190","title":"“Follow Your Interests Because Those Will Motivate You to Excel”: An Interview with Alister Cumming","year":2020,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Writing and Handwriting Education","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of British Columbia; York University","funders":"","keywords":"Psychology; Library science; Computer science","score_opus":0.07021481758870055,"score_gpt":0.39532355855434204,"score_spread":0.32510874096564146,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3111103222","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15303591,0.009134789,0.0016366924,0.79257447,0.0029450504,0.00033151457,0.00008030925,0.00007591745,0.040185332],"genre_scores_gemma":[0.69815856,0.010689803,0.0032847968,0.2359226,0.0008265948,0.0005607809,0.000054816803,0.00017424415,0.050327774],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9853087,0.0106632095,0.00027396344,0.00050616026,0.0016840092,0.0015639747],"domain_scores_gemma":[0.98099846,0.008803953,0.0010477647,0.0003574356,0.003015416,0.0057769935],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014246767,0.0008218489,0.000820258,0.0011686694,0.026111227,0.0066173957,0.0020343359,0.007346157,0.0030566351],"category_scores_gemma":[0.026531171,0.0011223546,0.0004714347,0.0012815618,0.012536152,0.010293255,0.0048590912,0.020047251,0.0010350252],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000029255894,0.000095362055,0.0018293438,0.00009508408,0.000008458416,0.0011064032,0.840051,0.000056693207,0.00054677227,0.0071771173,0.1378271,0.0111774],"study_design_scores_gemma":[0.000010570432,0.00006691457,0.0021252066,0.00030187075,0.000006219284,0.0007927716,0.783208,0.0001315388,0.00023155144,0.001368448,0.21168971,0.00006727795],"about_ca_topic_score_codex":0.05037967,"about_ca_topic_score_gemma":0.06769289,"teacher_disagreement_score":0.05037967,"about_ca_system_score_codex":0.008935649,"about_ca_system_score_gemma":0.008784625,"threshold_uncertainty_score":0.10017282},"labels":[],"label_agreement":null},{"id":"W3209416046","doi":"10.1080/15434303.2021.1992629","title":"The Relationship between Word Difficulty and Frequency: A Response to Hashimoto (2021)","year":2021,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Word lists by frequency; Correlation; Vocabulary; Rank (graph theory); Word (group theory); Linguistics; Psychology; Range (aeronautics); Mathematics; Statistics; Sentence; Philosophy","score_opus":0.01621099591469831,"score_gpt":0.3205802786497973,"score_spread":0.304369282735099,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3209416046","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11678341,0.0051522423,0.016411297,0.8313345,0.014884158,0.00027846915,0.00064881943,0.00025030202,0.014256749],"genre_scores_gemma":[0.4585101,0.0037971628,0.012588124,0.49230295,0.009501652,0.0005226714,0.00041154533,0.00023278738,0.02213305],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9949883,0.0019398951,0.0007243869,0.00072119845,0.0014138268,0.00021239452],"domain_scores_gemma":[0.9269078,0.05599312,0.0021626698,0.0014588896,0.011771689,0.0017058499],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011509527,0.0006895202,0.0005166484,0.001049753,0.0026294484,0.0016023528,0.0009629316,0.007035044,0.005616477],"category_scores_gemma":[0.08191672,0.00045225097,0.00066993636,0.0007401052,0.0018927384,0.0027221683,0.0025178432,0.00870917,0.0024284963],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011831387,0.0005507195,0.07132431,0.00081617065,0.0001333145,0.00290801,0.03837986,0.0010307401,0.013759308,0.03185965,0.62646866,0.21158625],"study_design_scores_gemma":[0.00020635458,0.0021158182,0.10981544,0.0011999169,0.00018328385,0.00880321,0.049387448,0.005274378,0.011625504,0.04394077,0.76671463,0.0007333368],"about_ca_topic_score_codex":0.0048644273,"about_ca_topic_score_gemma":0.0056924964,"teacher_disagreement_score":0.011509527,"about_ca_system_score_codex":0.0024014579,"about_ca_system_score_gemma":0.0014489327,"threshold_uncertainty_score":0.06086892},"labels":[],"label_agreement":null},{"id":"W4220982091","doi":"10.1080/15434303.2022.2038172","title":"Investigating the Effects of Task Type and Linguistic Background on Accuracy in Automated Speech Recognition Systems: Implications for Use in Language Assessment of Young Learners","year":2022,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Speech Recognition and Synthesis","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institute for Christian Studies; University of Toronto","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Task (project management); Computer science; Natural language processing; Artificial intelligence; Meaning (existential); Task analysis; Language proficiency; Test (biology); Psychology; Speech recognition","score_opus":0.03373161378752211,"score_gpt":0.33916975486964335,"score_spread":0.30543814108212125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4220982091","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9975151,0.00009934288,0.0015405599,0.00004908243,0.000007638987,0.00003830981,0.000029425748,0.000012744455,0.0007077893],"genre_scores_gemma":[0.99676204,0.00006673537,0.0025210814,0.000050725786,0.000014296887,0.00006319594,0.00007304451,0.000016400983,0.0004325242],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98045087,0.011827575,0.0014291393,0.0018505128,0.0039930767,0.00044880062],"domain_scores_gemma":[0.71413535,0.24428588,0.019041682,0.00793962,0.011002743,0.0035947177],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.027601201,0.00057076843,0.00061613834,0.0008754489,0.00052795577,0.0020893766,0.0006898741,0.0005519352,0.0013209978],"category_scores_gemma":[0.12513876,0.00033795962,0.0005875561,0.00069947215,0.00082292885,0.0018952234,0.0012771963,0.00068387704,0.0004997022],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002799876,0.0012154937,0.9232199,0.00012953869,0.00027382546,0.000120400444,0.0040192124,0.0010971172,0.01186227,0.00018945281,0.0001485276,0.054924257],"study_design_scores_gemma":[0.000028931625,0.003095093,0.9895056,0.00002535039,0.00005838739,0.00008529557,0.0010240942,0.0016761915,0.004172727,0.00015721371,0.00014743417,0.000023630897],"about_ca_topic_score_codex":0.0022451656,"about_ca_topic_score_gemma":0.0038311386,"teacher_disagreement_score":0.027601201,"about_ca_system_score_codex":0.0005326518,"about_ca_system_score_gemma":0.0007736636,"threshold_uncertainty_score":0.14597082},"labels":[],"label_agreement":null},{"id":"W4290958425","doi":"10.1080/15434303.2022.2073886","title":"Developing a Scenario-Based English Language Assessment in an Asian University","year":2022,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Language assessment; Linguistics; Language proficiency; Psychology; Mathematics education; Sociology; Computer science","score_opus":0.017125628697372018,"score_gpt":0.2795077938743099,"score_spread":0.2623821651769379,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4290958425","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.86926687,0.00007200305,0.09608491,0.001870965,0.00009420318,0.0034458304,0.00019617082,0.00059511734,0.028373871],"genre_scores_gemma":[0.81686985,0.00012725402,0.17742196,0.00017183216,0.000013881823,0.001024783,0.00024729464,0.000041762225,0.004081364],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9928288,0.0055548125,0.0003157376,0.00029886537,0.00059056893,0.0004112198],"domain_scores_gemma":[0.992303,0.0029020158,0.000468829,0.0004499121,0.002093159,0.001783057],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008776259,0.00070517475,0.0003047969,0.0008761144,0.0021146445,0.0037023863,0.0019462152,0.0013860277,0.0027196347],"category_scores_gemma":[0.012586098,0.0004562959,0.0004948117,0.00052202726,0.0011919456,0.003479391,0.003036566,0.001426995,0.0006677856],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009217536,0.021876609,0.102652565,0.0009320241,0.00018180913,0.010879377,0.24110857,0.09661947,0.047347486,0.050891906,0.021323768,0.40526465],"study_design_scores_gemma":[0.0005159819,0.012431721,0.055180207,0.001197118,0.00016602919,0.005988545,0.39048806,0.2635636,0.05645583,0.030272832,0.18271281,0.0010273169],"about_ca_topic_score_codex":0.0033190001,"about_ca_topic_score_gemma":0.008437269,"teacher_disagreement_score":0.008776259,"about_ca_system_score_codex":0.003035573,"about_ca_system_score_gemma":0.0055831145,"threshold_uncertainty_score":0.04641384},"labels":[],"label_agreement":null},{"id":"W4323544073","doi":"10.1080/15434303.2023.2184266","title":"Aligning Language Frameworks: An Example with the CLB and CEFR","year":2023,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Second Language Learning and Teaching","field":"Arts and Humanities","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Rasch model; Benchmarking; Dimension (graph theory); Computer science; German; Argument (complex analysis); Linguistics; Vocabulary; Certificate; Natural language processing; Computational linguistics; Language proficiency; Artificial intelligence; Psychology; Mathematics education","score_opus":0.020053347554890008,"score_gpt":0.2752021690098073,"score_spread":0.2551488214549173,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4323544073","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10690789,0.0013713357,0.48225152,0.013508979,0.0006634633,0.0012531468,0.0011050927,0.001488622,0.39145002],"genre_scores_gemma":[0.47706264,0.00042228404,0.50055766,0.0009912142,0.00004041865,0.00084944133,0.0010896038,0.00082999596,0.018156756],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.95760345,0.02317956,0.0015615837,0.0023668066,0.012546045,0.0027426428],"domain_scores_gemma":[0.9657239,0.011315338,0.0012534127,0.0062260837,0.014229566,0.0012515933],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.027012216,0.0007620074,0.00069201033,0.008431097,0.0068196836,0.008710169,0.0024423897,0.0022906011,0.00721273],"category_scores_gemma":[0.07585738,0.0005260843,0.0006020141,0.013049568,0.009718081,0.0068424847,0.008984303,0.004057551,0.0015124183],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000098612596,0.00010940954,0.007945474,0.00028336514,0.000019371437,0.0004960372,0.096450284,0.002067208,0.002496927,0.64780504,0.01061321,0.2316151],"study_design_scores_gemma":[0.0000437103,0.0001314777,0.020717949,0.0011411968,0.000042247466,0.0006152316,0.10756797,0.008786783,0.0056516426,0.16617577,0.6889187,0.0002072858],"about_ca_topic_score_codex":0.2938808,"about_ca_topic_score_gemma":0.3671863,"teacher_disagreement_score":0.2938808,"about_ca_system_score_codex":0.017487112,"about_ca_system_score_gemma":0.025392925,"threshold_uncertainty_score":0.58434045},"labels":[],"label_agreement":null},{"id":"W4385361752","doi":"10.1080/15434303.2023.2237487","title":"The Canadian English Language Proficiency Index Program (CELPIP) Test","year":2023,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Queen's University","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Index (typography); Language proficiency; Language assessment; Test (biology); Linguistics; Test of English as a Foreign Language; Psychology; Mathematics education; Computer science; Programming language","score_opus":0.012615323848987793,"score_gpt":0.2915964155539256,"score_spread":0.27898109170493784,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385361752","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.34863234,0.028103648,0.024337178,0.009043639,0.0015825451,0.00540474,0.061573766,0.0014631001,0.5198591],"genre_scores_gemma":[0.77714056,0.025302717,0.03788955,0.002628725,0.00017028769,0.0045434134,0.04140421,0.00024026234,0.110680364],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9972378,0.0001726049,0.00014018526,0.00016096311,0.0020504044,0.00023802715],"domain_scores_gemma":[0.9943698,0.00028830554,0.00022685803,0.00005772438,0.00470282,0.0003544819],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020099077,0.0005565409,0.0005348078,0.0027957922,0.0024182792,0.0013135474,0.0015532699,0.00039422465,0.007643822],"category_scores_gemma":[0.009605777,0.00018999637,0.0004907112,0.0030208474,0.0006340572,0.0008439224,0.0014697654,0.0009902286,0.0014490321],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002299983,0.00024469398,0.15867282,0.0012844796,0.00013048615,0.0005291351,0.0019599448,0.0010250863,0.0012730728,0.008984288,0.22331744,0.6023486],"study_design_scores_gemma":[0.00007133929,0.00025004253,0.6965152,0.0015006616,0.00017741395,0.000837019,0.0020560399,0.0017803751,0.0022274298,0.0018974122,0.29254842,0.0001387424],"about_ca_topic_score_codex":0.9039007,"about_ca_topic_score_gemma":0.9432771,"teacher_disagreement_score":0.09609932,"about_ca_system_score_codex":0.012620527,"about_ca_system_score_gemma":0.046330795,"threshold_uncertainty_score":0.19333047},"labels":[],"label_agreement":null},{"id":"W4389294437","doi":"10.1080/15434303.2023.2288253","title":"Validity Arguments for Automated Essay Scoring of Young Students’ Writing Traits","year":2023,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Text Readability and Simplification","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Writing assessment; Consistency (knowledge bases); Context (archaeology); Vocabulary; Psychology; Inference; Argument (complex analysis); Formative assessment; Trait; Artificial intelligence; Natural language processing; Computer science; Mathematics education; Linguistics","score_opus":0.0296680655231284,"score_gpt":0.359881211422216,"score_spread":0.3302131458990876,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389294437","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5678994,0.0022716634,0.36141697,0.019140787,0.0007077638,0.0018176963,0.0011721287,0.0008032663,0.044770274],"genre_scores_gemma":[0.94994336,0.00010477904,0.04696103,0.0007895456,0.00021512536,0.0006306105,0.00038789495,0.000093522096,0.0008741515],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.72553974,0.200425,0.0132478075,0.020165972,0.037793435,0.0028280213],"domain_scores_gemma":[0.13974434,0.7501418,0.027656032,0.048848253,0.032265577,0.001343967],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.23942912,0.0013941734,0.0015449704,0.005948508,0.0031656434,0.008760588,0.0048883506,0.0035908155,0.0028690596],"category_scores_gemma":[0.679516,0.0010802174,0.0022955025,0.0035894099,0.013440434,0.009080148,0.0070595625,0.004919419,0.0009716409],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004930359,0.0013135985,0.5719006,0.0009624443,0.0018286903,0.00050782855,0.013392254,0.028902525,0.0031112437,0.15395454,0.00586557,0.21333043],"study_design_scores_gemma":[0.0008671816,0.002453372,0.22070336,0.001611138,0.00068720547,0.0007309751,0.005711642,0.42391527,0.015743708,0.31606722,0.011143056,0.00036587624],"about_ca_topic_score_codex":0.004871405,"about_ca_topic_score_gemma":0.002731176,"teacher_disagreement_score":0.23942912,"about_ca_system_score_codex":0.00489859,"about_ca_system_score_gemma":0.004150737,"threshold_uncertainty_score":0.9379193},"labels":[],"label_agreement":null},{"id":"W4391927924","doi":"10.1080/15434303.2024.2311724","title":"The Development and Initial Validation of O-WSVLT, a Meaning-Recall Online L2 Spanish Vocabulary Levels Test","year":2024,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Second Language Acquisition and Learning","field":"Psychology","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Psychology; Meaning (existential); Vocabulary; Linguistics; Test (biology); Language proficiency; Recall; Vocabulary development; Cognitive psychology; Mathematics education","score_opus":0.023606083092814454,"score_gpt":0.3639875715569394,"score_spread":0.3403814884641249,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391927924","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9578837,0.00020531326,0.025272666,0.00044498488,0.00014387477,0.004666598,0.0018014645,0.00035741867,0.009223995],"genre_scores_gemma":[0.817521,0.0006287137,0.1446248,0.00061684434,0.000105037136,0.015455961,0.008366533,0.0003240116,0.012357036],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9951651,0.0017451429,0.0006832956,0.00059664814,0.00155044,0.00025935544],"domain_scores_gemma":[0.9872125,0.003936199,0.00084626366,0.0009403001,0.006373797,0.00069098687],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010784033,0.000613523,0.0005258482,0.0016500753,0.000438357,0.001153359,0.0010765012,0.00078190904,0.0020563488],"category_scores_gemma":[0.030843338,0.0002975633,0.0007330292,0.00066141057,0.0006329305,0.0012187776,0.0019045422,0.0010639019,0.0014691246],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008381261,0.0034912599,0.37200245,0.00047202632,0.00012621656,0.0006635051,0.009505229,0.0017133446,0.027257297,0.0020367545,0.005575188,0.5763185],"study_design_scores_gemma":[0.0006449897,0.011139042,0.8602297,0.0006257858,0.00018616981,0.002183216,0.010366538,0.010847922,0.040730603,0.0029922153,0.05985945,0.00019439685],"about_ca_topic_score_codex":0.0029144748,"about_ca_topic_score_gemma":0.004102418,"teacher_disagreement_score":0.010784033,"about_ca_system_score_codex":0.0007189761,"about_ca_system_score_gemma":0.003151255,"threshold_uncertainty_score":0.05703211},"labels":[],"label_agreement":null},{"id":"W4396694688","doi":"10.1080/15434303.2024.2346089","title":"The Academic Achievement of Undergraduate Students with Different English Language Proficiency Profiles","year":2024,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Second Language Learning and Teaching","field":"Arts and Humanities","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"York University","funders":"York University","keywords":"Language proficiency; Mathematics education; Academic achievement; Psychology; Language assessment","score_opus":0.009969094360222603,"score_gpt":0.28885734732641716,"score_spread":0.27888825296619457,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396694688","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9994936,0.000015890568,0.000026028052,0.0000137239185,0.0000014339613,0.0000033659442,0.000040322917,0.0000022484892,0.00040337746],"genre_scores_gemma":[0.9995409,0.00001652196,0.00003415327,0.000010706191,0.0000015311336,0.0000032982375,0.00010369046,0.0000010621137,0.00028812024],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9989478,0.00013330783,0.00010573714,0.00012278726,0.00033597555,0.0003544677],"domain_scores_gemma":[0.9974873,0.00031478168,0.0005242043,0.00010793238,0.0005973319,0.0009685363],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00109055,0.00027163926,0.00041238533,0.0021856646,0.0008759095,0.001517637,0.00045010055,0.00032597117,0.0019784002],"category_scores_gemma":[0.0055939923,0.00013255315,0.0003185017,0.0012516446,0.00069051713,0.00047002398,0.0013078768,0.00054226327,0.000625052],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005206433,0.00009643704,0.99266946,0.000004531668,0.000021810063,0.000060240713,0.0008478237,0.00003509369,0.00047418862,0.00005382949,0.00007515547,0.005609398],"study_design_scores_gemma":[0.0000016871097,0.000114295544,0.9979195,0.000002392017,0.000004532796,0.00008693885,0.0014474638,0.00007805488,0.00019716112,0.00003976735,0.00010413697,0.0000040522523],"about_ca_topic_score_codex":0.030207576,"about_ca_topic_score_gemma":0.04860394,"teacher_disagreement_score":0.030207576,"about_ca_system_score_codex":0.0011026169,"about_ca_system_score_gemma":0.0011036026,"threshold_uncertainty_score":0.06006348},"labels":[],"label_agreement":null},{"id":"W4399443232","doi":"10.1080/15434303.2024.2364172","title":"Adapting to a New Normal: A Review of <i>Technology Assisted Language Assessment in Diverse Contexts</i>","year":2024,"lang":"en","type":"review","venue":"Language Assessment Quarterly","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Language assessment; Psychology; Computer science; Linguistics; Mathematics education","score_opus":0.05299991408977252,"score_gpt":0.4712575397084361,"score_spread":0.4182576256186636,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399443232","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000034457003,0.9990388,0.000052403193,0.0005042003,0.00012768543,0.000004211006,0.000008700246,0.0000024888996,0.00022702965],"genre_scores_gemma":[0.0003908583,0.998686,0.00015258085,0.0005831255,0.000103143946,0.0000072030152,0.000014000759,0.0000011922205,0.0000618382],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99875045,0.00040961997,0.00030895858,0.00014239467,0.000333283,0.000055363336],"domain_scores_gemma":[0.992673,0.0054844916,0.00060403586,0.00010904893,0.00095232535,0.00017705388],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036235787,0.00072196656,0.0014336878,0.0040304675,0.00042358786,0.0017401229,0.0013476196,0.0015699337,0.0025711786],"category_scores_gemma":[0.009537996,0.0003648757,0.0010390155,0.004272188,0.00094296184,0.0026794053,0.0013219789,0.0024663813,0.0008076968],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000054998254,0.000039005292,0.0002392232,0.074978225,0.00023695514,0.000060363218,0.0001832929,0.0001535068,0.00028312363,0.002708842,0.029719517,0.8913429],"study_design_scores_gemma":[0.00003419376,0.00015614272,0.0024180019,0.11934771,0.00088438153,0.00087952707,0.00036615625,0.00011677922,0.0002218916,0.002698782,0.87282795,0.000048503145],"about_ca_topic_score_codex":0.003274708,"about_ca_topic_score_gemma":0.010708989,"teacher_disagreement_score":0.0040304675,"about_ca_system_score_codex":0.001091672,"about_ca_system_score_gemma":0.0048756488,"threshold_uncertainty_score":0.019163549},"labels":[],"label_agreement":null},{"id":"W4405148704","doi":"10.1080/15434303.2024.2438142","title":"The Academic Achievement of Undergraduate Students with Different TOEFL iBT Score Profiles: A Replication Study","year":2024,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Online Learning and Analytics","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"York University","funders":"York University","keywords":"Test of English as a Foreign Language; Replication (statistics); Psychology; Mathematics education; Academic achievement; Language assessment; Statistics; Mathematics","score_opus":0.012464814340156248,"score_gpt":0.349279442253663,"score_spread":0.33681462791350675,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405148704","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99814415,0.00002475589,0.00060662435,0.000034806482,0.000018508286,0.00026541026,0.00015757824,0.000025565487,0.000722535],"genre_scores_gemma":[0.9975048,0.000019290428,0.00096790225,0.000039361894,0.00000794648,0.00033301214,0.00033411855,0.000015231574,0.00077841914],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9959336,0.0015001469,0.000460659,0.0005692292,0.0010706906,0.0004657603],"domain_scores_gemma":[0.9766491,0.0030588699,0.0017124473,0.00826813,0.009028341,0.0012830831],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.009226554,0.0009591193,0.00080891105,0.0016081142,0.0015606967,0.001218094,0.0012747948,0.00079849403,0.0011271273],"category_scores_gemma":[0.032410685,0.00045588572,0.0014995256,0.0012056344,0.00094598485,0.0008798053,0.0014390313,0.0012723318,0.0007012258],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002148264,0.0040814616,0.8997283,0.00012444722,0.0004095348,0.0007087006,0.021755006,0.00042935458,0.0141186,0.00035645327,0.0012444799,0.05489529],"study_design_scores_gemma":[0.00019658596,0.0050833714,0.97670346,0.000031350643,0.00015140796,0.0005656211,0.008160716,0.0009838291,0.0055995267,0.00034468103,0.0021145549,0.000065020904],"about_ca_topic_score_codex":0.017200988,"about_ca_topic_score_gemma":0.016600922,"teacher_disagreement_score":0.99077344,"about_ca_system_score_codex":0.001287685,"about_ca_system_score_gemma":0.0019134732,"threshold_uncertainty_score":0.048795283},"labels":[],"label_agreement":null},{"id":"W4406605978","doi":"10.1080/15434303.2024.2448963","title":"Test Takers’ Attitudes Toward Varieties of Accents in Listening Tasks of the Duolingo English Test (2021 test version)","year":2025,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Linguistics, Language Diversity, and Identity","field":"Arts and Humanities","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Brock University","funders":"","keywords":"Test (biology); Active listening; Psychology; Communication","score_opus":0.010048740647190942,"score_gpt":0.2588661768107817,"score_spread":0.24881743616359078,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406605978","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9992341,0.000041815823,0.00007628677,0.00004264703,0.0000060419475,0.000010184343,0.000018181372,0.0000036215174,0.0005670688],"genre_scores_gemma":[0.99889994,0.0000687453,0.00015943157,0.00006999215,0.000007325918,0.00001925671,0.00006328724,0.000004533262,0.0007074396],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9967253,0.0009924544,0.00034894855,0.00034740884,0.0013362622,0.00024963237],"domain_scores_gemma":[0.9844096,0.005875305,0.004420427,0.0008490531,0.002636164,0.0018095487],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0060040313,0.00031732238,0.0003939106,0.0011547167,0.00048462613,0.0014940533,0.00034199824,0.00045791635,0.001507053],"category_scores_gemma":[0.022040566,0.0002370067,0.0005464428,0.00037100556,0.0008329727,0.0008539944,0.0015106515,0.00072915864,0.0005765598],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029703308,0.0002724123,0.9504811,0.00004528909,0.00007134487,0.00029437998,0.025581593,0.00009629679,0.00460243,0.000105865474,0.00039434983,0.017757908],"study_design_scores_gemma":[0.000019002357,0.0009716425,0.97039807,0.0000417154,0.000032535954,0.0005083285,0.024150254,0.00034270872,0.0019148234,0.00011132438,0.0014635769,0.00004600652],"about_ca_topic_score_codex":0.002746739,"about_ca_topic_score_gemma":0.0039453073,"teacher_disagreement_score":0.0060040313,"about_ca_system_score_codex":0.00039652738,"about_ca_system_score_gemma":0.00022296478,"threshold_uncertainty_score":0.031752765},"labels":[],"label_agreement":null},{"id":"W4406929912","doi":"10.1080/15434303.2025.2458599","title":"From Global Dependence to Local Expertise: An Interview with Rama Mathew","year":2025,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Education and Islamic Studies","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Psychology; Regional science; Economic geography; Sociology; Geography","score_opus":0.015369465846946137,"score_gpt":0.3964436286729267,"score_spread":0.38107416282598056,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406929912","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6482892,0.006406382,0.0011674576,0.27817956,0.0007752692,0.00015451267,0.00012891059,0.000067253684,0.06483151],"genre_scores_gemma":[0.943482,0.0043954444,0.0009912063,0.030100642,0.000101037,0.00014924271,0.00004773442,0.0000679965,0.020664813],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.99168456,0.0053133005,0.00019702184,0.0004339421,0.001153873,0.0012172634],"domain_scores_gemma":[0.992277,0.0038521544,0.000557069,0.00015737524,0.00081190385,0.0023444816],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0084247,0.0004892029,0.0008854762,0.0017811846,0.013395351,0.005066667,0.0019272161,0.003703032,0.0055017397],"category_scores_gemma":[0.011284589,0.0010715828,0.00033766427,0.0017663563,0.009872278,0.010635478,0.007924514,0.010976103,0.0007388294],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000017852986,0.00007095284,0.0021722743,0.000117193835,0.000006408499,0.0022041146,0.9585942,0.000049114613,0.000851331,0.0052032857,0.02373514,0.006978231],"study_design_scores_gemma":[0.0000025082043,0.000020701335,0.0016817785,0.00013388539,0.0000024869157,0.0005645019,0.95787495,0.00004750161,0.000063249674,0.0004488514,0.039138366,0.000021173688],"about_ca_topic_score_codex":0.048879933,"about_ca_topic_score_gemma":0.10190774,"teacher_disagreement_score":0.048879933,"about_ca_system_score_codex":0.0070459316,"about_ca_system_score_gemma":0.0056909425,"threshold_uncertainty_score":0.0971908},"labels":[],"label_agreement":null},{"id":"W4407098810","doi":"10.1080/15434303.2025.2455196","title":"Differential Item Functioning Due to Cultural Familiarity on a Large-Scale Reading Test: Does the Length of Residence Matter?","year":2025,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Differential item functioning; Psychology; Reading (process); Scale (ratio); Test (biology); Residence; Item response theory; Psychometrics; Developmental psychology; Linguistics; Geography; Sociology; Demography","score_opus":0.07156230965493464,"score_gpt":0.4314743672554329,"score_spread":0.35991205760049827,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407098810","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9977977,0.000119513694,0.0011088828,0.00008706714,0.00000988499,0.000029377044,0.000072658906,0.000009353273,0.0007656211],"genre_scores_gemma":[0.9987018,0.000041268635,0.0009784529,0.00004114277,0.000008308337,0.000019215398,0.00010174886,0.000006024228,0.00010208828],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99098575,0.0032054866,0.0010094615,0.00086563284,0.0034124162,0.00052121334],"domain_scores_gemma":[0.9428045,0.031206662,0.013448499,0.0037645805,0.007078586,0.0016971497],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014084324,0.0005070386,0.0005269156,0.0015444756,0.000669659,0.0012243079,0.0009580818,0.0005282943,0.0010989791],"category_scores_gemma":[0.06694912,0.00018433301,0.00081619545,0.0013772387,0.0013390859,0.0011117423,0.0010699052,0.00068158965,0.00024896156],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009315178,0.00003770415,0.9871761,0.000018469265,0.00007747139,0.000049173326,0.0007262327,0.0000811414,0.0004823262,0.000047784804,0.00006007154,0.011150504],"study_design_scores_gemma":[0.0000052769847,0.00017800443,0.99772066,0.000018583101,0.000032633427,0.00015410039,0.00075417804,0.00035079222,0.00051977695,0.00007934845,0.00017621876,0.000010315471],"about_ca_topic_score_codex":0.022635201,"about_ca_topic_score_gemma":0.06294477,"teacher_disagreement_score":0.022635201,"about_ca_system_score_codex":0.0013702204,"about_ca_system_score_gemma":0.0014962051,"threshold_uncertainty_score":0.0744859},"labels":[],"label_agreement":null},{"id":"W4410015948","doi":"10.1080/15434303.2025.2497818","title":"On the Interplay Between Conceptions of Assessment and Assessment Agency: Perspectives of Iranian EFL Preservice Teachers","year":2025,"lang":"en","type":"article","venue":"Language Assessment Quarterly","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Memorial University of Newfoundland","funders":"","keywords":"Mathematics education; Pedagogy; Agency (philosophy); Psychology; Alternative assessment; Semi-structured interview; Qualitative research; Sociology; Social science","score_opus":0.017476476578897192,"score_gpt":0.4119940254646767,"score_spread":0.39451754888577956,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410015948","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9864352,0.00075177074,0.0015077693,0.002659538,0.000014563854,0.000015027121,0.0000073560636,0.0000054992747,0.008603334],"genre_scores_gemma":[0.9994466,0.00013455217,0.00018432399,0.000056882473,0.0000026811463,0.0000033146628,0.000002025581,9.5177023e-7,0.00016860744],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.99212545,0.005459922,0.00024433678,0.00031403336,0.0011596536,0.0006966524],"domain_scores_gemma":[0.9842217,0.009286229,0.002746321,0.00043297262,0.0019214363,0.0013913497],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011378084,0.00030593402,0.0003282383,0.0016409938,0.0048369924,0.006694677,0.00077496516,0.0008716239,0.0006438482],"category_scores_gemma":[0.0132535,0.00036694252,0.00026809276,0.0009923446,0.014130337,0.002955193,0.0023682364,0.003114628,0.000068881396],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003718337,0.00010535364,0.0614876,0.00005197553,0.000011883828,0.00045875311,0.9081308,0.00024399676,0.0007051123,0.01719747,0.0003450235,0.011224681],"study_design_scores_gemma":[0.0000110157425,0.00008079194,0.038181707,0.000099386496,0.000013453051,0.0004755961,0.94642556,0.00070744083,0.00034560644,0.0063584126,0.0072631347,0.00003785331],"about_ca_topic_score_codex":0.02382147,"about_ca_topic_score_gemma":0.02462208,"teacher_disagreement_score":0.02382147,"about_ca_system_score_codex":0.006337504,"about_ca_system_score_gemma":0.0065972367,"threshold_uncertainty_score":0.06017375},"labels":[],"label_agreement":null}]}