{"meta":{"query_hash":"6e475cc8b99a","filters":{"venue":"Language Testing"},"cohort_total":57,"direct_labels_cover":0,"predictions_cover":57,"exported":57,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/6e475cc8b99a","api":"https://metacan.xera.ac/api/v1/cohort?venue=Language+Testing"},"results":[{"id":"W1979398480","doi":"10.1177/0265532214560799","title":"A prototype of a receptive lexical test for a polysynthetic heritage language: The case of Inuttitut in Labrador","year":2014,"lang":"en","type":"article","venue":"Language Testing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"The Scarborough Hospital; University of Toronto; Carleton University","funders":"","keywords":"Heritage language; Linguistics; Vocabulary; Psychology; Test (biology); Comprehension; Noun; Language proficiency; First language; Natural language processing; Computer science; Mathematics education","score_opus":0.016575572520713226,"score_gpt":0.2924684211711534,"score_spread":0.27589284865044017,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1979398480","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98232377,0.00013882674,0.0070403437,0.00051316933,0.000043658405,0.00033247203,0.00018174927,0.00023070442,0.0091953],"genre_scores_gemma":[0.94700766,0.00022151716,0.04564352,0.00035056076,0.000018414092,0.00038046678,0.00026329883,0.000089331246,0.006025229],"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99907374,0.0004099875,0.000086578126,0.00013665878,0.00010735896,0.00018569983],"domain_scores_gemma":[0.99891627,0.00037801373,0.0001078801,0.00015487739,0.00023612377,0.0002068478],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017867974,0.000557159,0.00033201723,0.0009145301,0.0012376657,0.0009160139,0.0013687074,0.0008671948,0.0030452062],"category_scores_gemma":[0.0028208995,0.0002486055,0.0002425395,0.00048984477,0.0013491941,0.0008349089,0.0011144966,0.0008388538,0.00118376],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00095408683,0.004352829,0.4051856,0.0005323555,0.00003182844,0.05747019,0.105017856,0.0011488503,0.0899139,0.004290867,0.005081096,0.32602054],"study_design_scores_gemma":[0.00027050602,0.008038996,0.5610843,0.0006933736,0.00011558376,0.0978569,0.14854996,0.0071944254,0.076599844,0.004925144,0.09430142,0.00036961696],"about_ca_topic_score_codex":0.017569259,"about_ca_topic_score_gemma":0.05738471,"teacher_disagreement_score":0.98243076,"about_ca_system_score_codex":0.0007900199,"about_ca_system_score_gemma":0.001944705,"threshold_uncertainty_score":0.034933984},"labels":[],"label_agreement":null},{"id":"W2005408543","doi":"10.1191/0265532204lt292oa","title":"Test decisions over time: tracking validity","year":2004,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":69,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Test (biology); Psychology; Sample (material); Active listening; Test validity; Applied psychology; Social psychology; Mathematics education; Psychometrics; Developmental psychology","score_opus":0.10042040920656546,"score_gpt":0.29390229656668887,"score_spread":0.1934818873601234,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2005408543","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95511067,0.00047931416,0.028086279,0.00079529034,0.00012131392,0.0010849611,0.00088304095,0.00016784994,0.01327132],"genre_scores_gemma":[0.9877666,0.00011285198,0.008922825,0.00013660392,0.000038142378,0.0010018806,0.00063416106,0.00005267294,0.001334183],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.90971655,0.04730969,0.008921941,0.0096258735,0.021771917,0.0026540274],"domain_scores_gemma":[0.45331055,0.3706812,0.08062076,0.05320383,0.039509382,0.0026743023],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.12248519,0.0005251965,0.00076955726,0.0059144855,0.0020404418,0.004325276,0.0026664953,0.001454551,0.0014094851],"category_scores_gemma":[0.4049832,0.00066029263,0.0013999502,0.0059001707,0.0037302775,0.005367981,0.0065129767,0.002184071,0.00052051427],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00033918006,0.00021178395,0.92107797,0.00012585787,0.0002026776,0.00007472829,0.011536235,0.0012168828,0.0005204959,0.0031013975,0.00044759255,0.06114526],"study_design_scores_gemma":[0.00008925429,0.0009019149,0.94728976,0.00027196668,0.00016494608,0.0002626238,0.0084279245,0.016031262,0.0041124397,0.012880387,0.00943786,0.00012967475],"about_ca_topic_score_codex":0.015717512,"about_ca_topic_score_gemma":0.0114666885,"teacher_disagreement_score":0.12248519,"about_ca_system_score_codex":0.003582884,"about_ca_system_score_gemma":0.004035691,"threshold_uncertainty_score":0.64777136},"labels":[],"label_agreement":null},{"id":"W2008716813","doi":"10.1177/02655322100270020902","title":"Book Review: Chapelle, C. A., Enright, M. K. and Jamieson, J. M. (Eds) Building a validity argument for the Test of English as a Foreign LanguageTM. New York and London: Routledge, Taylor &amp; Francis Group, 2008. 370 + xiii pp. ISBN 0-8058-5456-8 (paperback)","year":2010,"lang":"en","type":"article","venue":"Language Testing","topic":"Second Language Learning and Teaching","field":"Arts and Humanities","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Argument (complex analysis); Psychology; Media studies; Sociology; Chemistry","score_opus":0.02347074940022355,"score_gpt":0.2445379187758677,"score_spread":0.22106716937564413,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2008716813","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000084421044,0.96086943,0.00030406378,0.009209512,0.013818702,0.00006081664,0.00070523686,0.00008134317,0.014866547],"genre_scores_gemma":[0.0009859593,0.94988173,0.0005949036,0.004864792,0.004939995,0.0000870074,0.001018583,0.000053220956,0.03757383],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9989718,0.00009259829,0.00009994213,0.000114683164,0.0006481246,0.000072985604],"domain_scores_gemma":[0.99475765,0.0014683704,0.00045098164,0.00011617543,0.0028643552,0.0003424389],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012619507,0.0015753829,0.0032191821,0.0057726856,0.0007456938,0.0029022717,0.0031036143,0.0025728128,0.06754244],"category_scores_gemma":[0.005917161,0.0007761254,0.00093797524,0.0094930865,0.00091050105,0.0031649661,0.0012461103,0.0034285092,0.078203574],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000022035032,0.00001038975,0.000048642192,0.0037513755,0.000021179125,0.000047976246,0.000026633676,0.00003987522,0.00008178876,0.00035132287,0.86619633,0.12940246],"study_design_scores_gemma":[0.000015081661,0.000012887069,0.000315729,0.003173649,0.000036432717,0.00031952508,0.000028585948,0.000014703619,0.00005413507,0.00025223318,0.995765,0.000012004041],"about_ca_topic_score_codex":0.01106066,"about_ca_topic_score_gemma":0.021820642,"teacher_disagreement_score":0.06754244,"about_ca_system_score_codex":0.00212343,"about_ca_system_score_gemma":0.0046254196,"threshold_uncertainty_score":0.22595197},"labels":[],"label_agreement":null},{"id":"W2027399021","doi":"10.1177/0265532207076363","title":"The challenges of the Ontario Secondary School Literacy Test for second language students","year":2007,"lang":"en","type":"article","venue":"Language Testing","topic":"Second Language Acquisition and Learning","field":"Psychology","cited_by":57,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Queen's University","funders":"","keywords":"Literacy; Test (biology); Psychology; Mathematics education; Vocabulary; Context (archaeology); Reading (process); English as a second language; Language proficiency; Language assessment; Pedagogy; Linguistics","score_opus":0.02266960080524535,"score_gpt":0.3419876094318603,"score_spread":0.31931800862661497,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2027399021","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98352057,0.00021733418,0.00028104702,0.0005691026,0.00003835203,0.00010840475,0.00050884363,0.000035461984,0.014720854],"genre_scores_gemma":[0.9949885,0.000116640396,0.0006515359,0.00008861466,0.0000090116255,0.000075165844,0.000652091,0.000011555661,0.0034067465],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99767417,0.00025066384,0.00019108367,0.00014133188,0.0014406438,0.0003020704],"domain_scores_gemma":[0.9934663,0.0012068185,0.0008911825,0.00025735397,0.0027957857,0.0013825421],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017967555,0.00036620972,0.00047181058,0.0011362133,0.001441787,0.0015696998,0.0007051223,0.00042264746,0.0022106846],"category_scores_gemma":[0.014818118,0.00019566967,0.00037056263,0.0012203361,0.0006230456,0.000473904,0.0010899839,0.00048860686,0.00075553556],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026382285,0.0003008131,0.92818415,0.000059539892,0.000026878557,0.0003953329,0.0036458538,0.00018578423,0.0021519586,0.00045790084,0.0061419467,0.05818605],"study_design_scores_gemma":[0.000016343816,0.00012830594,0.99438053,0.000017755194,0.000007187971,0.00009296866,0.0011076032,0.0003258138,0.00045523533,0.000099740806,0.0033588146,0.000009631814],"about_ca_topic_score_codex":0.55166715,"about_ca_topic_score_gemma":0.7850741,"teacher_disagreement_score":0.99502563,"about_ca_system_score_codex":0.004974371,"about_ca_system_score_gemma":0.011111624,"threshold_uncertainty_score":0.9019463},"labels":[],"label_agreement":null},{"id":"W2042225150","doi":"10.1191/0265532204lt273oa","title":"Evaluation of an in-depth vocabulary knowledge measure for assessing reading performance","year":2004,"lang":"en","type":"article","venue":"Language Testing","topic":"Second Language Acquisition and Learning","field":"Psychology","cited_by":237,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Test of English as a Foreign Language; Vocabulary; Reading comprehension; Reading (process); Context (archaeology); Measure (data warehouse); Test (biology); Psychology; Language proficiency; Sample (material); Vocabulary development; Mathematics education; Language assessment; Computer science; Linguistics; Teaching method","score_opus":0.09731813955004215,"score_gpt":0.40605952944771595,"score_spread":0.3087413898976738,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2042225150","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9945739,0.00010233625,0.0025603874,0.000039320264,0.000010775397,0.00024564433,0.00017796396,0.00002517858,0.0022645968],"genre_scores_gemma":[0.98724115,0.00013663729,0.010833254,0.000037625134,0.000012691914,0.0003032234,0.00045384464,0.0000073557153,0.000974141],"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9951285,0.0013041709,0.0005165302,0.00030427382,0.0025223177,0.00022425925],"domain_scores_gemma":[0.9765419,0.010794364,0.004085685,0.0012000436,0.006073779,0.0013042025],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0052049537,0.00045869377,0.00043210786,0.0017843082,0.0003527467,0.0009910719,0.00089977257,0.0006165419,0.00080913503],"category_scores_gemma":[0.026027203,0.00017441408,0.0004490964,0.00073873013,0.00036721214,0.0012574405,0.0007220078,0.0006205713,0.0002609464],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007499933,0.0019559287,0.83119816,0.00019574637,0.00016124969,0.00019021303,0.0019545117,0.0017572062,0.01708511,0.00035117974,0.00055099366,0.14384963],"study_design_scores_gemma":[0.0000548773,0.0037903925,0.9796784,0.000046663565,0.00006580561,0.0003343201,0.0009486981,0.0054032737,0.0080637485,0.0001833622,0.0013951238,0.00003542263],"about_ca_topic_score_codex":0.009828704,"about_ca_topic_score_gemma":0.032676414,"teacher_disagreement_score":0.009828704,"about_ca_system_score_codex":0.0011211326,"about_ca_system_score_gemma":0.0015957336,"threshold_uncertainty_score":0.027526796},"labels":[],"label_agreement":null},{"id":"W2078562183","doi":"10.1177/0265532210384253","title":"Impact and consequences of school-based assessment (SBA): Students’ and parents’ views of SBA in Hong Kong","year":2011,"lang":"en","type":"article","venue":"Language Testing","topic":"Parental Involvement in Education","field":"Social Sciences","cited_by":92,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Perception; Psychology; Competence (human resources); Certificate; Context (archaeology); Medical education; Developmental psychology; Mathematics education; Pedagogy; Social psychology; Medicine","score_opus":0.2119144028794248,"score_gpt":0.4574731512778527,"score_spread":0.2455587483984279,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2078562183","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99958664,0.000034143563,0.000012614566,0.00004259589,0.0000018548623,0.0000024507194,0.0000104342525,6.3662463e-7,0.00030857918],"genre_scores_gemma":[0.99977046,0.000043325388,0.000020237936,0.000011883139,8.9511286e-7,0.0000029094513,0.000010765623,4.327843e-7,0.00013910358],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.99784875,0.0010140849,0.00026468546,0.00012130881,0.0004361683,0.0003150156],"domain_scores_gemma":[0.99208707,0.0019953114,0.002901582,0.00023336367,0.0011435519,0.0016392453],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028338083,0.0002204144,0.00026902376,0.000586298,0.00096811063,0.0016006563,0.00031423898,0.00029106214,0.0010924969],"category_scores_gemma":[0.0063637244,0.00027124071,0.00043494924,0.0005089076,0.00092127436,0.00058363134,0.0011826453,0.00070532627,0.00014590124],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007075116,0.00013704554,0.9627547,0.00003491764,0.000039137205,0.0005007365,0.029434726,0.0000865023,0.00045151656,0.000064656626,0.00014927455,0.006275937],"study_design_scores_gemma":[0.000003343356,0.00016346062,0.9519075,0.000024659546,0.000017820938,0.00012879736,0.04681305,0.00014116068,0.00022839686,0.000019306442,0.00053797005,0.000014489398],"about_ca_topic_score_codex":0.048288327,"about_ca_topic_score_gemma":0.061294347,"teacher_disagreement_score":0.048288327,"about_ca_system_score_codex":0.0014867767,"about_ca_system_score_gemma":0.0013637448,"threshold_uncertainty_score":0.0960145},"labels":[],"label_agreement":null},{"id":"W2082767250","doi":"10.1177/0265532208101010","title":"An investigation into native and non-native teachers' judgments of oral English performance: A mixed methods approach","year":2009,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":137,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Psychology; Pronunciation; Rasch model; Consistency (knowledge bases); Grammar; Mathematics education; First language; Internal consistency; Multimethodology; Linguistics; Psychometrics; Developmental psychology; Computer science; Artificial intelligence","score_opus":0.04154220340015909,"score_gpt":0.3200407438186342,"score_spread":0.2784985404184751,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2082767250","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98794484,0.00016721556,0.009387945,0.00007053011,0.000019203977,0.00093060563,0.00007954667,0.000016855382,0.0013832542],"genre_scores_gemma":[0.9609865,0.00022902407,0.033741903,0.00017243021,0.000022918564,0.0033989227,0.00012088211,0.000021556667,0.0013058294],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9665099,0.024289452,0.0022471102,0.0019282856,0.004367743,0.0006575187],"domain_scores_gemma":[0.93909216,0.044640586,0.004326257,0.003399726,0.0076838788,0.0008573094],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.041703705,0.00064800656,0.00085434725,0.0031159103,0.0026031258,0.0029566824,0.0013061234,0.0006082811,0.0010263015],"category_scores_gemma":[0.05004807,0.0006764524,0.00058158895,0.0016199735,0.0016998346,0.0013668395,0.0018598904,0.0005619381,0.00026819375],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013902712,0.003530941,0.3214507,0.000972909,0.0003686852,0.0008487219,0.47385812,0.00048383742,0.017294414,0.0015844926,0.00028131308,0.17793564],"study_design_scores_gemma":[0.0004271279,0.0130927265,0.38849658,0.0005438776,0.0004753736,0.0014547174,0.5517167,0.0056561385,0.026214054,0.0025887564,0.009041518,0.00029242638],"about_ca_topic_score_codex":0.0045669223,"about_ca_topic_score_gemma":0.015367123,"teacher_disagreement_score":0.041703705,"about_ca_system_score_codex":0.0016276458,"about_ca_system_score_gemma":0.0022548318,"threshold_uncertainty_score":0.22055286},"labels":[],"label_agreement":null},{"id":"W2096343999","doi":"10.1191/0265532203lt248oa","title":"Does item-level DIF manifest itself in scale-level analyses? Implications for translating language tests","year":2003,"lang":"en","type":"article","venue":"Language Testing","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":92,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Differential item functioning; Equivalence (formal languages); Psychology; Scale (ratio); Measurement invariance; Item response theory; Item analysis; Statistics; Test (biology); Psychometrics; Developmental psychology; Linguistics; Structural equation modeling; Mathematics; Confirmatory factor analysis","score_opus":0.6773012489253456,"score_gpt":0.5390986227521983,"score_spread":0.13820262617314727,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2096343999","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4390763,0.0037870647,0.4727606,0.048561856,0.001222171,0.0015856741,0.0006147871,0.0013630118,0.031028483],"genre_scores_gemma":[0.85194045,0.00066388387,0.1421973,0.0028258192,0.00028537313,0.0008046106,0.00017754335,0.0002399375,0.00086511177],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.75313956,0.21547058,0.009110659,0.005530565,0.01497175,0.0017768658],"domain_scores_gemma":[0.38939604,0.5342252,0.017157307,0.03871311,0.019060813,0.0014475093],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.17416789,0.0016680086,0.0025087902,0.0037579997,0.0021000307,0.0066151386,0.003071767,0.002287974,0.0038271842],"category_scores_gemma":[0.665731,0.0009926538,0.0023001507,0.004827591,0.011322169,0.011017878,0.003984601,0.0041018436,0.0009048782],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021170257,0.00095115916,0.2104652,0.0023710018,0.0007993317,0.0029976442,0.04360043,0.018668082,0.0055897185,0.15426247,0.0054606157,0.5527174],"study_design_scores_gemma":[0.00068931264,0.0024450917,0.13222495,0.002315832,0.00046343985,0.0023018434,0.04180646,0.09224727,0.015492978,0.69716394,0.0124541195,0.00039470094],"about_ca_topic_score_codex":0.0028548748,"about_ca_topic_score_gemma":0.0025772378,"teacher_disagreement_score":0.17416789,"about_ca_system_score_codex":0.0036113758,"about_ca_system_score_gemma":0.0037287788,"threshold_uncertainty_score":0.9210988},"labels":[],"label_agreement":null},{"id":"W2097625210","doi":"10.1177/0265532211404195","title":"Test review: ACCESS for ELLs <sup>®</sup>","year":2011,"lang":"en","type":"article","venue":"Language Testing","topic":"Digital Accessibility for Disabilities","field":"Social Sciences","cited_by":16,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Ell; Psychology; Test (biology); Language proficiency; Mathematics education; Teaching method; Geology","score_opus":0.13373330771807518,"score_gpt":0.3811891609412508,"score_spread":0.24745585322317565,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2097625210","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0035091161,0.8056568,0.0015628783,0.09603358,0.043002248,0.0014099122,0.00451857,0.0003201432,0.043986827],"genre_scores_gemma":[0.035437047,0.7901912,0.0047572902,0.06598925,0.017495012,0.0021117376,0.008913721,0.00026194484,0.07484279],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99294454,0.0025189843,0.001380812,0.00029211378,0.002514786,0.00034891363],"domain_scores_gemma":[0.9455424,0.02070571,0.005420247,0.0008102643,0.024898965,0.0026224128],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008904257,0.0010939382,0.0022579536,0.0048078015,0.00090077746,0.0025965602,0.0025255266,0.004017615,0.030048719],"category_scores_gemma":[0.055561036,0.0003270379,0.001467699,0.0038798943,0.0011784195,0.0024259281,0.0016832063,0.0017027411,0.009522787],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038124732,0.00008989713,0.00068220205,0.023837158,0.00016916134,0.0001814362,0.000088371075,0.00006076592,0.0004976707,0.0008145637,0.60076034,0.37243706],"study_design_scores_gemma":[0.00033180974,0.0006418225,0.00772302,0.052909993,0.0011784282,0.0007828777,0.00031277243,0.00013090238,0.0012033965,0.0010505765,0.9336714,0.00006294154],"about_ca_topic_score_codex":0.012878567,"about_ca_topic_score_gemma":0.028794246,"teacher_disagreement_score":0.030048719,"about_ca_system_score_codex":0.0031270087,"about_ca_system_score_gemma":0.011137778,"threshold_uncertainty_score":0.100522995},"labels":[],"label_agreement":null},{"id":"W2100317885","doi":"10.1177/0265532215570924","title":"How do young students with different profiles of reading skill mastery, perceived ability, and goal orientation respond to holistic diagnostic feedback?","year":2015,"lang":"en","type":"article","venue":"Language Testing","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":61,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Psychology; Reading (process); Goal orientation; Cognition; Developmental psychology; Orientation (vector space); Mathematics education; Cognitive psychology; Social psychology","score_opus":0.06663303109817183,"score_gpt":0.3901718268217531,"score_spread":0.3235387957235813,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2100317885","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9996716,0.000045531415,0.00006525503,0.000019993427,0.0000016895931,0.0000032144405,0.000015284966,0.0000018932022,0.00017549683],"genre_scores_gemma":[0.99954516,0.000069488546,0.00009472314,0.000029620607,0.0000016104822,0.0000055839473,0.00004685948,0.0000016587608,0.00020517116],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99957556,0.00007421904,0.00004452172,0.00009334303,0.00012432397,0.000088035806],"domain_scores_gemma":[0.99749255,0.000545133,0.0011849238,0.00011257134,0.00028438686,0.00038032723],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009951954,0.00016845154,0.00032430165,0.00075741275,0.00019620416,0.0013203972,0.00021928058,0.00052246655,0.0008466254],"category_scores_gemma":[0.005394971,0.00021461479,0.00032883685,0.00036022908,0.0003536002,0.00096579274,0.0004940955,0.0004194899,0.0004206041],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008681572,0.00011994014,0.97709686,0.00002816981,0.000034373716,0.00011387609,0.0047083846,0.000058170706,0.0025889352,0.000039449482,0.00007450683,0.015050509],"study_design_scores_gemma":[0.0000031264672,0.00013722148,0.9938287,0.000010480811,0.000011133519,0.00015156041,0.0048723784,0.00014911793,0.000601974,0.000069900365,0.00015549688,0.000008960192],"about_ca_topic_score_codex":0.002819832,"about_ca_topic_score_gemma":0.0043703257,"teacher_disagreement_score":0.002819832,"about_ca_system_score_codex":0.00028660503,"about_ca_system_score_gemma":0.00030400112,"threshold_uncertainty_score":0.00560683},"labels":[],"label_agreement":null},{"id":"W2104209819","doi":"10.1177/0265532212436659","title":"Topical knowledge and ESL writing","year":2012,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":77,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of British Columbia","funders":"","keywords":"Impromptu; Cohesion (chemistry); Language proficiency; Psychology; Test (biology); Mathematics education; Language assessment; Second language writing; Test of English as a Foreign Language; English for academic purposes; Task (project management); Pedagogy; Second language; Linguistics; Computer science","score_opus":0.061547222700138966,"score_gpt":0.29217058559962256,"score_spread":0.2306233628994836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2104209819","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99303716,0.00020001444,0.00041369023,0.00007272478,0.0000062274817,0.0000100093375,0.000026538006,0.000017031913,0.0062165135],"genre_scores_gemma":[0.9978923,0.0001324444,0.00039984894,0.000019513152,0.00000897995,0.000009183703,0.00004872256,0.000008146099,0.0014809008],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9973273,0.0007975129,0.00020764527,0.0003107834,0.0011699852,0.00018670908],"domain_scores_gemma":[0.94951856,0.030754382,0.0111277895,0.0019071433,0.003928151,0.0027640124],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017523984,0.00030343072,0.00035464356,0.0011746974,0.0005090172,0.0019282273,0.00039443892,0.0003298881,0.00416008],"category_scores_gemma":[0.042420156,0.00013033347,0.0001709647,0.0006123751,0.00110405,0.0010705921,0.001222518,0.0006481987,0.000538443],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00084325374,0.0022312964,0.7010803,0.00026937877,0.00012375845,0.001686136,0.012763438,0.00074395415,0.020715559,0.00086836366,0.00096543756,0.25770906],"study_design_scores_gemma":[0.000026606553,0.0009517952,0.98318696,0.0000791755,0.000053428586,0.0009848257,0.0055111796,0.00082446926,0.0049808193,0.001491023,0.0018764877,0.000033294844],"about_ca_topic_score_codex":0.0014374425,"about_ca_topic_score_gemma":0.00199298,"teacher_disagreement_score":0.00416008,"about_ca_system_score_codex":0.00052286783,"about_ca_system_score_gemma":0.0005677603,"threshold_uncertainty_score":0.01391685},"labels":[],"label_agreement":null},{"id":"W2105451781","doi":"10.1191/0265532204lt287oa","title":"Teacher formative assessment and talk in classroom contexts: assessment as discourse and assessment of discourse","year":2004,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":159,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Formative assessment; Psychology; Pedagogy; Discourse analysis; Applied linguistics; Mathematics education; Assessment for learning; Systemic functional linguistics; Linguistics","score_opus":0.030320027060730614,"score_gpt":0.42869087904920095,"score_spread":0.39837085198847033,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2105451781","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5576613,0.010088329,0.35316706,0.0066699083,0.00032928615,0.0008942871,0.00017751243,0.00038709395,0.070625216],"genre_scores_gemma":[0.95153934,0.001755937,0.043193083,0.00017051023,0.00008937758,0.00065605034,0.00004003309,0.000050146988,0.0025054733],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.95229363,0.038467374,0.0016987859,0.0012080843,0.005936554,0.00039564478],"domain_scores_gemma":[0.9081358,0.07414794,0.007252437,0.00433499,0.004460025,0.0016688736],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.034785565,0.000617564,0.0007696394,0.005578764,0.0014768598,0.01191711,0.0016777656,0.0014989567,0.0016222023],"category_scores_gemma":[0.10054566,0.00035868637,0.00035311715,0.0036866933,0.013881913,0.009962572,0.0053009973,0.0023104127,0.00028052373],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019708356,0.00047035806,0.051817276,0.0014294807,0.00009859348,0.000492739,0.4369698,0.0021914595,0.0061090277,0.12878841,0.0010665038,0.37036923],"study_design_scores_gemma":[0.00012887265,0.0018416012,0.16395037,0.0035380274,0.00015519871,0.004283031,0.3164106,0.020575251,0.01895963,0.40266913,0.06704799,0.00044031028],"about_ca_topic_score_codex":0.0015557848,"about_ca_topic_score_gemma":0.002265697,"teacher_disagreement_score":0.034785565,"about_ca_system_score_codex":0.0022964398,"about_ca_system_score_gemma":0.0033879909,"threshold_uncertainty_score":0.18396586},"labels":[],"label_agreement":null},{"id":"W2109863812","doi":"10.1191/0265532204lt278oa","title":"A teacher-verification study of speaking and writing prototype tasks for a new TOEFL","year":2004,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":93,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Test of English as a Foreign Language; Active listening; Psychology; Formative assessment; Mathematics education; Test (biology); CLARITY; Reading (process); Second language writing; Language proficiency; Presentation (obstetrics); Language assessment; Pedagogy; Computer science; Second language; Linguistics; Communication","score_opus":0.08046825576428888,"score_gpt":0.31140587081200116,"score_spread":0.23093761504771226,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2109863812","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99863905,0.000017137567,0.0007689169,0.000027168646,0.000008696666,0.000121520374,0.000015089303,0.000015538253,0.00038687454],"genre_scores_gemma":[0.9916943,0.000060485447,0.0055071004,0.0001148041,0.000016144633,0.0003316253,0.0001041072,0.00002439609,0.0021471211],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.992465,0.004347793,0.0007357041,0.0007294015,0.0013268987,0.00039522877],"domain_scores_gemma":[0.8910272,0.07368827,0.006842709,0.009079237,0.016200839,0.0031617335],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018614706,0.0006208867,0.0009603996,0.001292594,0.0018530082,0.0017354285,0.0015650115,0.0009866271,0.0017019438],"category_scores_gemma":[0.08789999,0.0007758045,0.00048413186,0.00055526715,0.0011569269,0.0018612868,0.0012331499,0.0018890181,0.0007459284],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028559384,0.03294211,0.23885012,0.00083844777,0.00007237328,0.0026155207,0.44075152,0.001073167,0.044423956,0.00064287765,0.0020232373,0.23291063],"study_design_scores_gemma":[0.0018267004,0.05766236,0.61407197,0.00039747462,0.00014089353,0.0052338615,0.22544582,0.0105841365,0.053408824,0.001231481,0.029559772,0.0004366987],"about_ca_topic_score_codex":0.0032719078,"about_ca_topic_score_gemma":0.009383092,"teacher_disagreement_score":0.018614706,"about_ca_system_score_codex":0.0016311512,"about_ca_system_score_gemma":0.0018221535,"threshold_uncertainty_score":0.09844518},"labels":[],"label_agreement":null},{"id":"W2116070328","doi":"10.1177/0265532210376379","title":"Think-aloud protocols in research on essay rating: An empirical study of their veridicality and reactivity","year":2010,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":121,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Think aloud protocol; Psychology; Protocol analysis; Perception; Empirical research; Sample (material); Qualitative research; Rating scale; Social psychology; Nomothetic and idiographic; Cognitive psychology; Applied psychology; Developmental psychology; Epistemology; Cognitive science","score_opus":0.29237641940546794,"score_gpt":0.5510859877479262,"score_spread":0.25870956834245823,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2116070328","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.89397186,0.00041844405,0.09813229,0.00019223375,0.00012734091,0.0013186177,0.00013550199,0.00023610612,0.005467661],"genre_scores_gemma":[0.913073,0.00052427937,0.07989421,0.00026898735,0.000116369105,0.0036343818,0.00019491986,0.0001571192,0.0021368158],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9395154,0.047427915,0.0033638806,0.0029153635,0.006367211,0.00041019727],"domain_scores_gemma":[0.6944738,0.2487216,0.022762796,0.017106093,0.015704732,0.001230987],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.037438773,0.00073611207,0.00049278885,0.0012075497,0.00091735413,0.0016428737,0.0011022746,0.00094669766,0.00116799],"category_scores_gemma":[0.21999006,0.000555056,0.00032053198,0.0010708567,0.001426186,0.0015625042,0.0019572917,0.0012809291,0.00066098967],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0029120299,0.001885879,0.11458012,0.0022114736,0.00035529106,0.0007919685,0.22303063,0.001853537,0.11902476,0.0054787006,0.0017866923,0.526089],"study_design_scores_gemma":[0.00074646063,0.020407619,0.5165713,0.0025433183,0.00057559216,0.007781459,0.123501666,0.027272105,0.2092096,0.0289061,0.06154568,0.00093917776],"about_ca_topic_score_codex":0.00019331435,"about_ca_topic_score_gemma":0.00026656708,"teacher_disagreement_score":0.96256125,"about_ca_system_score_codex":0.0004902453,"about_ca_system_score_gemma":0.0006471811,"threshold_uncertainty_score":0.19799751},"labels":[],"label_agreement":null},{"id":"W2122362059","doi":"10.1191/0265532206lt322oa","title":"Aiming for positive washback: a case study of international teaching assistants","year":2005,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":102,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université Laval","funders":"","keywords":"Test (biology); Psychology; Language proficiency; Process (computing); Mathematics education; Empirical research; Computer science","score_opus":0.05418180584152639,"score_gpt":0.417552964658704,"score_spread":0.36337115881717763,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2122362059","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9960716,0.00008090044,0.001216433,0.0005325318,0.00001895767,0.00013888397,0.000011370151,0.000016232561,0.0019131047],"genre_scores_gemma":[0.99296796,0.00025939735,0.0033303946,0.0003558495,0.000033881788,0.00014890042,0.00001667783,0.000019655314,0.002867276],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.99234456,0.0044353264,0.0003190832,0.00044943244,0.000847509,0.0016041644],"domain_scores_gemma":[0.98438764,0.008667849,0.0017862826,0.00075169635,0.0010980531,0.0033084718],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007090907,0.0009132704,0.0006940333,0.0011775739,0.0074764076,0.002904533,0.0025957774,0.003783207,0.00242489],"category_scores_gemma":[0.026065875,0.0007124443,0.0006381013,0.0009941871,0.0026886323,0.0016174798,0.003125132,0.0041075842,0.00054787344],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006586729,0.015612577,0.09269043,0.0005907764,0.00008675403,0.09080338,0.6864872,0.001088879,0.006645918,0.002685027,0.001792372,0.100858055],"study_design_scores_gemma":[0.00017531101,0.006551826,0.04126441,0.00032412398,0.00007951962,0.033497617,0.8831282,0.0023797809,0.008789723,0.0016023105,0.022056464,0.00015069722],"about_ca_topic_score_codex":0.004929525,"about_ca_topic_score_gemma":0.013003497,"teacher_disagreement_score":0.0074764076,"about_ca_system_score_codex":0.0024064893,"about_ca_system_score_gemma":0.0027270336,"threshold_uncertainty_score":0.0375008},"labels":[],"label_agreement":null},{"id":"W2123012719","doi":"10.1177/026553220101800302","title":"Examining dialogue: another approach to content specification and to validating inferences drawn from test scores","year":2001,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":162,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Construct (python library); Psychology; Test (biology); Sociocultural evolution; Point (geometry); Cognition; Cognitive psychology; Content (measure theory); Mathematics education; Inference; Construct validity; Linguistics; Natural language processing; Social psychology; Computer science; Artificial intelligence; Psychometrics; Developmental psychology","score_opus":0.24408084356114834,"score_gpt":0.27518337385510755,"score_spread":0.031102530293959207,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2123012719","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12087889,0.00036422894,0.84470457,0.002643687,0.00028124233,0.0028102396,0.0012719521,0.001939338,0.025105888],"genre_scores_gemma":[0.33061194,0.00017012769,0.65932876,0.0008451003,0.00018413231,0.0040144753,0.0011918945,0.00043173737,0.0032218755],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.78463984,0.15800653,0.017984757,0.0113404,0.025277993,0.0027504282],"domain_scores_gemma":[0.34976852,0.49968186,0.032480165,0.052570634,0.062539235,0.0029596556],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.16165599,0.002729091,0.002763749,0.027303629,0.0036116983,0.015623417,0.005862653,0.0043938966,0.0041745673],"category_scores_gemma":[0.40974957,0.00080237404,0.0022233923,0.015087536,0.009089683,0.016914358,0.009106664,0.0051659076,0.0015367124],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00109646,0.0015009136,0.112669386,0.0024299538,0.0007175228,0.0010875316,0.10910203,0.0051302053,0.018363861,0.13031873,0.0048890617,0.61269444],"study_design_scores_gemma":[0.00057415955,0.0052031293,0.12797916,0.0033057965,0.0010046263,0.0021552749,0.11094303,0.09504624,0.0782926,0.45944402,0.11471744,0.0013344862],"about_ca_topic_score_codex":0.0053348895,"about_ca_topic_score_gemma":0.0039410265,"teacher_disagreement_score":0.16165599,"about_ca_system_score_codex":0.0037531971,"about_ca_system_score_gemma":0.0064892257,"threshold_uncertainty_score":0.85492885},"labels":[],"label_agreement":null},{"id":"W2129988987","doi":"10.1177/0265532208092433","title":"Test review: College English Test (CET) in China","year":2008,"lang":"en","type":"article","venue":"Language Testing","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":192,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Test (biology); College English; China; Psychology; Language assessment; Test of English as a Foreign Language; Mathematics education; Political science","score_opus":0.017274515028303867,"score_gpt":0.2757766721564041,"score_spread":0.25850215712810026,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2129988987","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01718771,0.9630475,0.0004798873,0.007553943,0.004514816,0.00047637126,0.002155648,0.00003385757,0.0045503527],"genre_scores_gemma":[0.16098055,0.81126505,0.0015038842,0.013534357,0.0032081208,0.0010002333,0.004036045,0.00004609154,0.0044255815],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9954639,0.00139658,0.001563624,0.000360683,0.0010829276,0.00013232726],"domain_scores_gemma":[0.97349936,0.01147348,0.0035064928,0.00047578677,0.01022287,0.00082204596],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0074461414,0.0007602825,0.0032256502,0.0045355354,0.0005979651,0.0011928958,0.0015942514,0.0011918185,0.0029008812],"category_scores_gemma":[0.030593777,0.00031881427,0.0010489933,0.006943128,0.00088431226,0.00094920985,0.0007716336,0.0006229384,0.00046994266],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015786885,0.0002392376,0.036948,0.15162778,0.0028294108,0.0009439738,0.00047810373,0.00042837593,0.0011365076,0.0007585466,0.1393269,0.6637045],"study_design_scores_gemma":[0.0018530022,0.0033641183,0.31416893,0.1312393,0.021050053,0.0030938834,0.0015650834,0.0008385925,0.0024010558,0.00089003495,0.5193116,0.00022432236],"about_ca_topic_score_codex":0.040743962,"about_ca_topic_score_gemma":0.0809317,"teacher_disagreement_score":0.040743962,"about_ca_system_score_codex":0.0030941218,"about_ca_system_score_gemma":0.01092561,"threshold_uncertainty_score":0.08101356},"labels":[],"label_agreement":null},{"id":"W2131222006","doi":"10.1177/0265532210364380","title":"Use of tree-based regression in the analyses of L2 reading test items","year":2010,"lang":"en","type":"article","venue":"Language Testing","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Reading (process); Cognition; Psychology; Interpretation (philosophy); Cognitive psychology; Test (biology); Tree (set theory); Regression; Regression analysis; Natural language processing; Artificial intelligence; Computer science; Machine learning; Linguistics; Mathematics","score_opus":0.10655832357545793,"score_gpt":0.3862878005927542,"score_spread":0.2797294770172963,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2131222006","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15426277,0.00023764661,0.8382488,0.00017392485,0.000108935026,0.0008729113,0.00073740067,0.0027836033,0.002574039],"genre_scores_gemma":[0.5544535,0.00024379669,0.44033542,0.00009071834,0.00005335671,0.001299517,0.001198814,0.0010568344,0.0012680695],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9493062,0.044806585,0.0010845137,0.0021204476,0.0022436988,0.00043854583],"domain_scores_gemma":[0.78412205,0.1924188,0.007942911,0.0077794716,0.0072387764,0.000498054],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04071942,0.002715764,0.001815071,0.0042362516,0.0006670157,0.001988568,0.0014181634,0.0007584753,0.0023252966],"category_scores_gemma":[0.16624518,0.00065829896,0.002386888,0.0056106485,0.0005987318,0.0026882577,0.0013493816,0.0027308695,0.00166329],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0021892658,0.0010380433,0.22031212,0.0011300951,0.0031484563,0.0006095589,0.0053842426,0.10603649,0.012329335,0.017206514,0.005826845,0.624789],"study_design_scores_gemma":[0.00017266127,0.0018387499,0.07227311,0.0002717504,0.0007602227,0.00058860035,0.0011463541,0.89564747,0.006311596,0.014785732,0.005971241,0.00023250029],"about_ca_topic_score_codex":0.0069377148,"about_ca_topic_score_gemma":0.0062781554,"teacher_disagreement_score":0.04071942,"about_ca_system_score_codex":0.0007155774,"about_ca_system_score_gemma":0.0012301075,"threshold_uncertainty_score":0.21534747},"labels":[],"label_agreement":null},{"id":"W2134278381","doi":"10.1177/0265532207083743","title":"The key to success: English language testing in China","year":2008,"lang":"en","type":"article","venue":"Language Testing","topic":"Multilingual Education and Policy","field":"Social Sciences","cited_by":242,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"Ministry of Education, India; Ministry of Earth Sciences","keywords":"Language assessment; China; Context (archaeology); Psychology; English language; Test (biology); Linguistics; Test of English as a Foreign Language; Language proficiency; Chinese language; Mathematics education; History","score_opus":0.0648421376261999,"score_gpt":0.4101903691881555,"score_spread":0.3453482315619556,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2134278381","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6194819,0.01485494,0.0026721766,0.21254838,0.0006924016,0.00023033691,0.00024262852,0.00011617389,0.14916106],"genre_scores_gemma":[0.98831546,0.0029132664,0.0006792504,0.0032718459,0.000121647776,0.000055953144,0.00006687543,0.0000110154415,0.0045647076],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99294186,0.002335205,0.00049750315,0.0004724337,0.0022890228,0.0014640193],"domain_scores_gemma":[0.97786134,0.008931503,0.0031785192,0.0007986427,0.0039055943,0.0053243707],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010966676,0.0002848776,0.00041121393,0.0021448263,0.003949618,0.005014567,0.0010941307,0.0012882241,0.00429056],"category_scores_gemma":[0.023207353,0.00021766995,0.00018871088,0.0043643666,0.008342939,0.004767886,0.0037331334,0.0019985046,0.00033952633],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015301148,0.00032692592,0.34621993,0.0012871013,0.00004655093,0.003073614,0.041900452,0.0013017231,0.001824679,0.19462012,0.030072046,0.37917387],"study_design_scores_gemma":[0.000060005485,0.00044837492,0.7214758,0.001610371,0.00007416737,0.00090928585,0.058187354,0.0028051203,0.0032041327,0.051487517,0.15954661,0.0001913222],"about_ca_topic_score_codex":0.08723613,"about_ca_topic_score_gemma":0.06559477,"teacher_disagreement_score":0.08723613,"about_ca_system_score_codex":0.0074636093,"about_ca_system_score_gemma":0.03821543,"threshold_uncertainty_score":0.17345673},"labels":[],"label_agreement":null},{"id":"W2135663628","doi":"10.1177/0265532208097336","title":"Cognitive diagnostic assessment of L2 reading comprehension ability: Validity arguments for Fusion Model application to <i>LanguEdge</i> assessment","year":2008,"lang":"en","type":"article","venue":"Language Testing","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":175,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Reading comprehension; Psychology; Cognition; Test (biology); Profiling (computer programming); Dependability; Comprehension; Reading (process); Cognitive psychology; Computer science; Linguistics","score_opus":0.48241198306348876,"score_gpt":0.5159648486704076,"score_spread":0.033552865606918836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2135663628","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.74994373,0.00089486875,0.22588971,0.005164768,0.00016942859,0.00069538294,0.00045545018,0.000531794,0.016254878],"genre_scores_gemma":[0.96534127,0.00006738762,0.03382828,0.00020369959,0.00003523198,0.00022431156,0.00009728307,0.000019579744,0.00018295908],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.96500546,0.020720027,0.0019154969,0.0028489777,0.008765366,0.0007446035],"domain_scores_gemma":[0.7637811,0.19121632,0.008795157,0.0173636,0.016994955,0.0018489063],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.048682097,0.0007586219,0.0010727313,0.0043581612,0.00096537324,0.0038033433,0.002019352,0.0017546772,0.002137555],"category_scores_gemma":[0.25082552,0.00034591265,0.0013667243,0.0021922865,0.004630828,0.0049365256,0.0052087675,0.0022575702,0.00037939433],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0034204125,0.00086839637,0.45910537,0.0006854163,0.00077974796,0.0006137151,0.010717166,0.020135134,0.007388768,0.072537534,0.0032128694,0.42053553],"study_design_scores_gemma":[0.0003997541,0.0022024757,0.1764929,0.0005914565,0.0005031759,0.0014374626,0.005373914,0.60010755,0.015892472,0.19187413,0.0047699017,0.00035482072],"about_ca_topic_score_codex":0.0033312219,"about_ca_topic_score_gemma":0.0016661857,"teacher_disagreement_score":0.048682097,"about_ca_system_score_codex":0.0023677102,"about_ca_system_score_gemma":0.0021957702,"threshold_uncertainty_score":0.25745857},"labels":[],"label_agreement":null},{"id":"W2136852860","doi":"10.1177/026553220101800303","title":"Native- and nonnative-speaking EFL teachers’ evaluation of Chinese students’ English writing","year":2001,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":117,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Psychology; Multivariate analysis of variance; First language; Foreign language; English as a foreign language; Point (geometry); Language assessment; Language proficiency; Linguistics; Mathematics education","score_opus":0.06659189121337757,"score_gpt":0.33968104876054345,"score_spread":0.2730891575471659,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2136852860","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9995974,0.000013718931,0.000025722702,0.0000042368874,0.0000012383878,0.0000039168253,0.0000055911296,0.0000012403818,0.00034700328],"genre_scores_gemma":[0.9992281,0.000030065505,0.000081554295,0.000007730022,0.000002611948,0.00001089476,0.000028374516,0.000001664512,0.0006089127],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9990682,0.0002329998,0.00014813547,0.00010419183,0.0003455208,0.000100912155],"domain_scores_gemma":[0.9929945,0.0021279461,0.0015640828,0.0003610742,0.0018077357,0.0011447375],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019739086,0.00029276288,0.00028800723,0.00083909166,0.00055978214,0.0006305817,0.00019197234,0.00022219423,0.0015877427],"category_scores_gemma":[0.009412385,0.00013832186,0.0002044293,0.00031725765,0.0005758541,0.0003604811,0.00062077364,0.00024655892,0.00036413557],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009907834,0.0005527464,0.8243179,0.00022504892,0.00010223757,0.0007832336,0.07450287,0.00027949488,0.048594587,0.0000966666,0.0003880839,0.049166396],"study_design_scores_gemma":[0.0000319511,0.0006257896,0.974772,0.00001409444,0.000014496788,0.00032945676,0.019725742,0.000351763,0.0034640278,0.000042288302,0.0006046671,0.00002365589],"about_ca_topic_score_codex":0.0033808742,"about_ca_topic_score_gemma":0.0077884733,"teacher_disagreement_score":0.0033808742,"about_ca_system_score_codex":0.00044305183,"about_ca_system_score_gemma":0.0004686286,"threshold_uncertainty_score":0.0104391575},"labels":[],"label_agreement":null},{"id":"W2137984677","doi":"10.1177/0265532207071510","title":"A confirmatory approach to differential item functioning on an ESL reading assessment","year":2006,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":57,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Alberta Advanced Education","funders":"","keywords":"Differential item functioning; Psychology; Reading (process); Test (biology); Language proficiency; Item response theory; Flagging; Cognitive psychology; Psychometrics; Developmental psychology; Linguistics; Mathematics education","score_opus":0.04018946173988295,"score_gpt":0.27198873090135683,"score_spread":0.2317992691614739,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2137984677","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18895237,0.00054675527,0.7808871,0.0018418662,0.00029171392,0.0053907954,0.0009454471,0.00095068116,0.02019334],"genre_scores_gemma":[0.5546777,0.00020232823,0.4369658,0.00057719136,0.00009771049,0.005045694,0.00084963086,0.00011627434,0.001467605],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9245896,0.0522566,0.0052260407,0.005923046,0.011005518,0.0009991938],"domain_scores_gemma":[0.7379405,0.13970426,0.015761303,0.035857424,0.06920537,0.0015310865],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.09328806,0.0025263785,0.0014458975,0.011437383,0.0025732499,0.0029258775,0.0018708629,0.0010225889,0.002979851],"category_scores_gemma":[0.21079837,0.00068340136,0.001965057,0.0048000664,0.0022236228,0.0027508251,0.003253985,0.0027228312,0.0010023563],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009849996,0.0010358886,0.33873257,0.0017209174,0.0009467673,0.0010172545,0.038044807,0.0030444742,0.016804757,0.076645434,0.006040409,0.51498175],"study_design_scores_gemma":[0.0008131514,0.006222638,0.46995762,0.00280176,0.002017147,0.004852135,0.047376968,0.11503453,0.042144183,0.26245436,0.045472257,0.00085320644],"about_ca_topic_score_codex":0.0067551173,"about_ca_topic_score_gemma":0.0094355,"teacher_disagreement_score":0.09328806,"about_ca_system_score_codex":0.0019602177,"about_ca_system_score_gemma":0.0050235637,"threshold_uncertainty_score":0.49336028},"labels":[],"label_agreement":null},{"id":"W2145055698","doi":"10.1177/0265532210368717","title":"Explaining ESL essay holistic scores: A multilevel modeling approach","year":2010,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":64,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Educational Testing Service","keywords":"Psychology; Argumentation theory; Context (archaeology); Multilevel model; Set (abstract data type); Multilevel modelling; Social psychology; Epistemology; Statistics; Computer science","score_opus":0.11955947065560128,"score_gpt":0.29601648065408587,"score_spread":0.17645700999848457,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2145055698","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.76555216,0.00025073683,0.22892831,0.00065238384,0.00006473838,0.000614025,0.0014296919,0.00037415337,0.0021338845],"genre_scores_gemma":[0.9431535,0.00006960533,0.054540806,0.000043033488,0.000026406058,0.00073400786,0.00068767654,0.000053124826,0.000691813],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9846848,0.011020328,0.0006814038,0.001720024,0.0013865868,0.0005068664],"domain_scores_gemma":[0.9512467,0.037976887,0.0038238165,0.0037781019,0.0026731992,0.0005012836],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021438377,0.0011499575,0.0012520392,0.0033859422,0.0012768504,0.0025000514,0.0017552436,0.00089326175,0.003217856],"category_scores_gemma":[0.05999744,0.00055656745,0.0031972695,0.0029715425,0.0007871265,0.0013548617,0.0027149622,0.0021105714,0.00051269995],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044578902,0.00043049207,0.8650311,0.00024046886,0.0028346872,0.00033035595,0.0073590237,0.019431919,0.0019551911,0.011329178,0.001978673,0.08863313],"study_design_scores_gemma":[0.00010454703,0.001206616,0.41788724,0.00021167213,0.0011079794,0.0003594108,0.0033120343,0.5430786,0.0021646325,0.025901716,0.0044748317,0.0001907043],"about_ca_topic_score_codex":0.012747144,"about_ca_topic_score_gemma":0.009672363,"teacher_disagreement_score":0.021438377,"about_ca_system_score_codex":0.0012194099,"about_ca_system_score_gemma":0.0012447229,"threshold_uncertainty_score":0.113378346},"labels":[],"label_agreement":null},{"id":"W2145523583","doi":"10.1177/026553220101800206","title":"ESL/EFL instructors' practices for writing assessment: specific purposes or general purposes?","year":2001,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":57,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Interview; Psychology; Mathematics education; English for academic purposes; Pedagogy; Writing assessment; Language assessment; Process (computing); Sociology; Computer science","score_opus":0.14286572258107863,"score_gpt":0.3633483596418732,"score_spread":0.22048263706079455,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2145523583","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97547865,0.00039750908,0.01141316,0.0023278657,0.00005320139,0.00015128547,0.00002671045,0.00006892504,0.010082726],"genre_scores_gemma":[0.99277,0.00018011025,0.004997498,0.000271631,0.000016822538,0.00014136556,0.000018449708,0.00002392397,0.0015801595],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9851059,0.009808451,0.0011393175,0.0009636215,0.0020744242,0.00090818986],"domain_scores_gemma":[0.9687103,0.015769469,0.0047081625,0.0024289899,0.0059704925,0.0024125318],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01512459,0.000282616,0.00039075283,0.0011887159,0.0017875265,0.0032574728,0.0010288253,0.0008897464,0.00096649956],"category_scores_gemma":[0.056308385,0.00031893895,0.00022310195,0.0010108212,0.0029415258,0.0032747923,0.002297257,0.0016169401,0.00035831006],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000054604054,0.00019561438,0.10680001,0.00027518944,0.000014788402,0.0005099837,0.77563334,0.00015710355,0.004662677,0.004665891,0.0014438173,0.10558694],"study_design_scores_gemma":[0.000037260193,0.00037626005,0.15309411,0.0007481105,0.000026243537,0.001507744,0.7893734,0.0019095712,0.0039159968,0.005431462,0.04346956,0.000110344095],"about_ca_topic_score_codex":0.0015784537,"about_ca_topic_score_gemma":0.0037876503,"teacher_disagreement_score":0.01512459,"about_ca_system_score_codex":0.0013912002,"about_ca_system_score_gemma":0.0022678669,"threshold_uncertainty_score":0.07998741},"labels":[],"label_agreement":null},{"id":"W2155682078","doi":"10.1191/0265532206lt337oa","title":"How assessing reading comprehension with multiple-choice questions shapes the construct: a cognitive processing perspective","year":2006,"lang":"en","type":"article","venue":"Language Testing","topic":"Reading and Literacy Development","field":"Psychology","cited_by":243,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Construct (python library); Reading comprehension; Psychology; Variety (cybernetics); Comprehension; Cognitive psychology; Cognition; Multiple choice; Perspective (graphical); Test (biology); Reading (process); Selection (genetic algorithm); Task (project management); Computer science; Linguistics; Artificial intelligence","score_opus":0.02646085463337262,"score_gpt":0.3216621472035859,"score_spread":0.29520129257021327,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2155682078","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7053012,0.0016538183,0.26722062,0.004521044,0.000060190534,0.00021557961,0.00009362078,0.00024206423,0.020691797],"genre_scores_gemma":[0.9506222,0.00054492336,0.04762226,0.0004274304,0.00004773659,0.00016216816,0.000047080386,0.00005740225,0.00046880473],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9756699,0.018742999,0.00059530145,0.0017730257,0.0028359161,0.00038280577],"domain_scores_gemma":[0.81436425,0.168916,0.007079685,0.005206371,0.0038854454,0.00054838334],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.027349968,0.00064954936,0.0005401817,0.0032951368,0.0005193373,0.0070097675,0.0012035164,0.001346379,0.0013596124],"category_scores_gemma":[0.110720515,0.0005837109,0.0006254528,0.0018709619,0.009179497,0.009850972,0.0022261676,0.001989328,0.00023817808],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048502252,0.0008221492,0.13668865,0.0015504402,0.00037914998,0.0007410812,0.28942478,0.0061696162,0.036600303,0.17012122,0.001154344,0.35586324],"study_design_scores_gemma":[0.00017988673,0.0016162377,0.24529083,0.0010309384,0.0003266637,0.0022672848,0.07645,0.046807025,0.033853978,0.57387674,0.017766027,0.00053431356],"about_ca_topic_score_codex":0.0012488419,"about_ca_topic_score_gemma":0.0014256394,"teacher_disagreement_score":0.027349968,"about_ca_system_score_codex":0.0012154533,"about_ca_system_score_gemma":0.00076363387,"threshold_uncertainty_score":0.14464217},"labels":[],"label_agreement":null},{"id":"W2156770411","doi":"10.1177/0265532209104666","title":"Interacting in pairs in a test of oral proficiency: Co-constructing a better performance","year":2009,"lang":"en","type":"article","venue":"Language Testing","topic":"Multilingual Education and Policy","field":"Social Sciences","cited_by":163,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Test (biology); Psychology; Meaning (existential); Context (archaeology); Negotiation; Social psychology; Mathematics education","score_opus":0.06423787690980926,"score_gpt":0.44141249514360664,"score_spread":0.3771746182337974,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2156770411","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99831045,0.000026960019,0.00043580701,0.00005518669,0.0000059011545,0.000012468237,0.000012697901,0.000010458704,0.001129929],"genre_scores_gemma":[0.9974848,0.000031609805,0.0013891427,0.000024120833,0.0000051254983,0.000017109598,0.00003400016,0.000009182167,0.001004926],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99004304,0.00591205,0.0005085882,0.0008335784,0.0019947975,0.0007079415],"domain_scores_gemma":[0.9813172,0.0074005807,0.004617434,0.001873934,0.0026490001,0.0021418969],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0068014376,0.00056830584,0.00064899545,0.0014439374,0.0014265826,0.00430192,0.00076833984,0.00062348455,0.0025798585],"category_scores_gemma":[0.03441317,0.00038517048,0.0004515966,0.0005162649,0.0019632978,0.0014135708,0.0038504514,0.00094462535,0.0009158096],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038481472,0.0009792591,0.77584934,0.00010031735,0.000099233286,0.0014651029,0.1321388,0.00037929337,0.012675371,0.0004649543,0.0005889723,0.074874565],"study_design_scores_gemma":[0.000028247601,0.0021729874,0.8458275,0.000052540356,0.00007729335,0.002120726,0.13588539,0.0017434714,0.0076940963,0.00094938936,0.0032857212,0.0001625235],"about_ca_topic_score_codex":0.0029514509,"about_ca_topic_score_gemma":0.004255696,"teacher_disagreement_score":0.0068014376,"about_ca_system_score_codex":0.0007872855,"about_ca_system_score_gemma":0.0009887612,"threshold_uncertainty_score":0.035969913},"labels":[],"label_agreement":null},{"id":"W2159609851","doi":"10.1191/0265532204lt288oa","title":"ESL/EFL instructors’ classroom assessment practices: purposes, methods, and procedures","year":2004,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":184,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Alberta; Queen's University","funders":"","keywords":"Psychology; English as a foreign language; Mathematics education; Tertiary level; Pedagogy; Second language; Language assessment; Linguistics","score_opus":0.06401151606627827,"score_gpt":0.4566933338742858,"score_spread":0.39268181780800754,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2159609851","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.96071553,0.00033067845,0.01445856,0.0004667196,0.000037044312,0.007025859,0.0004268307,0.00023839943,0.016300356],"genre_scores_gemma":[0.9315376,0.00036200185,0.052951995,0.00024309706,0.00003468316,0.010535528,0.00025868588,0.000052633823,0.00402367],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.95475256,0.027016846,0.005813431,0.0028335128,0.007963849,0.001619732],"domain_scores_gemma":[0.9352966,0.022130534,0.006743305,0.0064598476,0.026848633,0.0025210904],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.044195637,0.00059095735,0.0005557169,0.004196423,0.0024874099,0.0023115845,0.0014129665,0.00053172756,0.0020146067],"category_scores_gemma":[0.073291585,0.00041830871,0.0002716429,0.0029839056,0.002312854,0.0013039681,0.0031252482,0.0007570963,0.001132535],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00088822603,0.002255439,0.2514513,0.0007885541,0.000029856768,0.00036994877,0.15963951,0.00046729107,0.0112481965,0.0010812179,0.0019613898,0.5698192],"study_design_scores_gemma":[0.00032431594,0.002467649,0.7964455,0.0010251944,0.00006302998,0.00054443744,0.123649575,0.0030743196,0.030028025,0.001700874,0.040468313,0.00020877109],"about_ca_topic_score_codex":0.01155767,"about_ca_topic_score_gemma":0.023019541,"teacher_disagreement_score":0.044195637,"about_ca_system_score_codex":0.004144968,"about_ca_system_score_gemma":0.0062302044,"threshold_uncertainty_score":0.23373169},"labels":[],"label_agreement":null},{"id":"W2162394022","doi":"10.1177/0265532212469178","title":"Differential importance of language components in determining secondary school students’ Chinese reading literacy performance","year":2013,"lang":"en","type":"article","venue":"Language Testing","topic":"Reading and Literacy Development","field":"Psychology","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Dictation; Psychology; Reading (process); Copying; Reading comprehension; Literacy; Mathematics education; Chinese characters; Linguistics; Pedagogy; Computer science; Artificial intelligence","score_opus":0.018120185307201827,"score_gpt":0.3170660889351399,"score_spread":0.29894590362793805,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2162394022","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9997149,0.000013219315,0.000017675386,0.0000038718035,7.4376635e-7,0.0000037340064,0.000028073744,0.0000014643205,0.0002163331],"genre_scores_gemma":[0.9997008,0.000013328679,0.00004022146,0.0000036844558,8.0817824e-7,0.000004426587,0.000070362345,9.9218e-7,0.00016536278],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99944776,0.00009321092,0.00009418823,0.00012161136,0.00013028468,0.000112842135],"domain_scores_gemma":[0.99820936,0.00042076365,0.0005813711,0.00011224409,0.00032977414,0.0003465138],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00067908235,0.00043850832,0.00030748968,0.0013680911,0.00037226878,0.0007151695,0.00023307429,0.0002591197,0.001379534],"category_scores_gemma":[0.0025915906,0.00020943792,0.00029259853,0.000856646,0.00052587275,0.00037915766,0.00053248834,0.00030634325,0.00040344134],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00004234772,0.00003505239,0.9940037,0.000010316536,0.000012976675,0.00010230909,0.0010880942,0.000017540035,0.0014129161,0.000015967671,0.000021114045,0.0032376735],"study_design_scores_gemma":[9.509521e-7,0.000042175878,0.9993067,0.0000012460847,0.0000044687717,0.00004084109,0.00036001374,0.000040016097,0.000157042,0.000007419055,0.00003770011,0.0000014843997],"about_ca_topic_score_codex":0.01442404,"about_ca_topic_score_gemma":0.023837158,"teacher_disagreement_score":0.01442404,"about_ca_system_score_codex":0.0003608872,"about_ca_system_score_gemma":0.00050759397,"threshold_uncertainty_score":0.028680146},"labels":[],"label_agreement":null},{"id":"W2168715918","doi":"10.1177/02655322090260020602","title":"Test review: The Versant Spanish <sup>TM</sup> Test","year":2009,"lang":"en","type":"article","venue":"Language Testing","topic":"Neurobiology of Language and Bilingualism","field":"Neuroscience","cited_by":8,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Test (biology); Psychology; Geology","score_opus":0.03386165576883628,"score_gpt":0.2910663115460739,"score_spread":0.2572046557772376,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2168715918","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032664265,0.5737682,0.008308944,0.119863145,0.06584161,0.00067142205,0.006571662,0.0017276779,0.19058318],"genre_scores_gemma":[0.1655442,0.4158945,0.013663536,0.08308638,0.032354746,0.0006834267,0.013893669,0.0015721909,0.27330735],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988005,0.0003370694,0.00014993697,0.000116889685,0.0005280547,0.00006755192],"domain_scores_gemma":[0.9925505,0.0019404301,0.0004243231,0.00034354182,0.004209574,0.0005315853],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024637943,0.00073737657,0.0011917538,0.0013612846,0.0004766186,0.0010760126,0.0019258069,0.0021134245,0.01560763],"category_scores_gemma":[0.010081322,0.00018408886,0.00046851984,0.0011096552,0.0007596905,0.0009967093,0.00066310645,0.0011639206,0.009289597],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00048304422,0.00012565566,0.002446895,0.0014433172,0.000077681776,0.00085137726,0.00003094462,0.00009008203,0.0014113645,0.0011095167,0.42618507,0.5657451],"study_design_scores_gemma":[0.00016430125,0.0005448613,0.010221398,0.0013722464,0.00015777448,0.0041332953,0.000083895815,0.00016530885,0.0023558715,0.0012183016,0.979548,0.000034786568],"about_ca_topic_score_codex":0.00697942,"about_ca_topic_score_gemma":0.017303841,"teacher_disagreement_score":0.01560763,"about_ca_system_score_codex":0.0012861553,"about_ca_system_score_gemma":0.00331052,"threshold_uncertainty_score":0.052212715},"labels":[],"label_agreement":null},{"id":"W2318200755","doi":"10.1177/0265532213509810","title":"Examining the impact of L2 proficiency and keyboarding skills on scores on TOEFL-iBT writing tasks","year":2013,"lang":"en","type":"article","venue":"Language Testing","topic":"Writing and Handwriting Education","field":"Social Sciences","cited_by":45,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Test of English as a Foreign Language; Psychology; Test (biology); Language proficiency; Task (project management); Context (archaeology); English language; Mathematics education","score_opus":0.04433657734822507,"score_gpt":0.3557343872788868,"score_spread":0.31139780993066174,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2318200755","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99936694,0.000026899283,0.000060097198,0.000011095617,0.00000250757,0.000006077299,0.00007609324,0.000006527842,0.00044375355],"genre_scores_gemma":[0.9990152,0.000017551152,0.00009335104,0.000010737381,0.0000041621456,0.000012792001,0.00017592586,0.0000057094476,0.00066451903],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9978167,0.0005957266,0.00024859954,0.0003454314,0.00071583857,0.00027770465],"domain_scores_gemma":[0.9737083,0.014618672,0.005825583,0.0013373543,0.0023485194,0.002161499],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025948752,0.00077478756,0.0005740676,0.00100547,0.0003168082,0.0010427142,0.0005009238,0.00044784683,0.0031486645],"category_scores_gemma":[0.017750917,0.00021471386,0.0008072724,0.0005511574,0.0006130955,0.00074615487,0.0010222845,0.00077703083,0.00076485414],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008459972,0.00053596223,0.9800465,0.000036953683,0.00019093508,0.00026928447,0.0011713684,0.0002386138,0.0039657657,0.00002139372,0.0001934804,0.012483817],"study_design_scores_gemma":[0.000006489735,0.00074493006,0.9979868,0.000003269268,0.000017207676,0.00007127903,0.00023067681,0.00016525318,0.00068749196,0.000007853972,0.00007297423,0.0000057492575],"about_ca_topic_score_codex":0.0042784605,"about_ca_topic_score_gemma":0.0051950603,"teacher_disagreement_score":0.0042784605,"about_ca_system_score_codex":0.0003866562,"about_ca_system_score_gemma":0.00027756483,"threshold_uncertainty_score":0.013723195},"labels":[],"label_agreement":null},{"id":"W2337150116","doi":"10.1177/0265532214559115","title":"Design in four diagnostic language assessments","year":2015,"lang":"en","type":"article","venue":"Language Testing","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Language assessment; Language proficiency; Psychology; Educational assessment; Process (computing); Intervention (counseling); Educational research; Mathematics education; Management science; Computer science; Pedagogy","score_opus":0.2169898390405153,"score_gpt":0.45664737825958907,"score_spread":0.23965753921907376,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2337150116","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.39462498,0.0032200853,0.4009633,0.0063343514,0.0014756452,0.03884285,0.0014414215,0.0024137183,0.15068361],"genre_scores_gemma":[0.48841554,0.0007269115,0.46821922,0.0018402536,0.00008664049,0.022732126,0.00076936575,0.000248319,0.01696162],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9487635,0.032579754,0.006099351,0.0041372534,0.0066334396,0.0017868436],"domain_scores_gemma":[0.9171156,0.04341727,0.006101175,0.010321442,0.019408245,0.0036361958],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04240267,0.0009022532,0.0007321286,0.0042251307,0.0028806482,0.0065779733,0.003126456,0.0022314694,0.007169635],"category_scores_gemma":[0.09704426,0.0009262599,0.00093273906,0.0024832792,0.0041818884,0.0037820963,0.008320687,0.0023058578,0.0022146357],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024089986,0.0023231686,0.05155953,0.0032104729,0.00013308106,0.00085772254,0.0805931,0.0033905148,0.008093774,0.11512407,0.012378131,0.7199275],"study_design_scores_gemma":[0.002325702,0.005836597,0.059715647,0.004425993,0.0006106213,0.0024874995,0.0804157,0.016071567,0.0347132,0.16867086,0.6241461,0.0005804698],"about_ca_topic_score_codex":0.0017045886,"about_ca_topic_score_gemma":0.0028818743,"teacher_disagreement_score":0.04240267,"about_ca_system_score_codex":0.0065436712,"about_ca_system_score_gemma":0.011349418,"threshold_uncertainty_score":0.22424942},"labels":[],"label_agreement":null},{"id":"W2599956837","doi":"10.1177/0265532216684576","title":"Book Review: Focus on Assessment","year":2016,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Focus (optics); Psychology; Linguistics; Philosophy; Optics","score_opus":0.032919987551438844,"score_gpt":0.3820371374063785,"score_spread":0.3491171498549397,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2599956837","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00021572,0.86540973,0.0005078069,0.033716016,0.08324899,0.000108859436,0.000244589,0.000077404984,0.016470943],"genre_scores_gemma":[0.0025707837,0.7831941,0.00084395515,0.035481606,0.096836634,0.00029123394,0.00043766867,0.00014676683,0.08019734],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9963728,0.0009885344,0.0003695254,0.00034101724,0.001734142,0.00019400126],"domain_scores_gemma":[0.97439945,0.011693326,0.0018304056,0.00043226936,0.009765799,0.0018787518],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0038767129,0.0013775115,0.0032618681,0.009181574,0.0006649477,0.0050927415,0.0016451323,0.0038212894,0.03993967],"category_scores_gemma":[0.026358377,0.0005778097,0.00095743715,0.009497337,0.0012991398,0.0033399882,0.0019784754,0.0045454754,0.022344206],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002834032,0.000023874594,0.00007753744,0.0040723905,0.000023076294,0.000046329184,0.000024362247,0.00005936294,0.00010631386,0.00060117914,0.8857649,0.10917237],"study_design_scores_gemma":[0.000033321416,0.00004624589,0.0006890746,0.006838445,0.000048471655,0.00036695026,0.000052432406,0.00003632399,0.000057437366,0.0006401936,0.9911747,0.000016378799],"about_ca_topic_score_codex":0.0034027696,"about_ca_topic_score_gemma":0.013135827,"teacher_disagreement_score":0.03993967,"about_ca_system_score_codex":0.0033571783,"about_ca_system_score_gemma":0.0055536265,"threshold_uncertainty_score":0.1336115},"labels":[],"label_agreement":null},{"id":"W2611914815","doi":"10.1177/0265532217703433","title":"Developing a user-oriented second language comprehensibility scale for English-medium universities","year":2017,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":59,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Alberta; Concordia University","funders":"","keywords":"Operationalization; Formative assessment; Psychology; Construct (python library); English for academic purposes; Scale (ratio); Language proficiency; Point (geometry); Task (project management); Focus (optics); Rating scale; Mathematics education; Linguistics; Computer science","score_opus":0.04907756014366304,"score_gpt":0.29045579136247907,"score_spread":0.24137823121881602,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2611914815","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95698214,0.00010317702,0.029787308,0.00032272155,0.00009236213,0.004272165,0.00071722263,0.00034954207,0.007373238],"genre_scores_gemma":[0.8324381,0.00022455941,0.15175007,0.00015851633,0.000030493644,0.01059772,0.0016824603,0.00008426433,0.003033924],"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9965264,0.0010463748,0.0006784603,0.0002134978,0.001347456,0.00018776146],"domain_scores_gemma":[0.98024803,0.009440136,0.0018520966,0.00094509457,0.0066388273,0.0008756489],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0117777595,0.00055275677,0.0005555551,0.002307403,0.00062745024,0.0020271577,0.0011172232,0.00080933125,0.0022197699],"category_scores_gemma":[0.032071617,0.00047580624,0.0013739177,0.0007732368,0.0005265892,0.002107647,0.0022958324,0.0016283233,0.0009028001],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00060565694,0.0021794718,0.5337923,0.0009707609,0.0002595792,0.00057711883,0.05831706,0.002789708,0.029148106,0.0045168754,0.00966571,0.35717764],"study_design_scores_gemma":[0.00015010679,0.0031439003,0.91505826,0.00052464363,0.000091417794,0.0006516168,0.021776421,0.014358906,0.012185832,0.004114476,0.027676763,0.00026771196],"about_ca_topic_score_codex":0.00091306644,"about_ca_topic_score_gemma":0.0015929003,"teacher_disagreement_score":0.0117777595,"about_ca_system_score_codex":0.0008873254,"about_ca_system_score_gemma":0.0017266949,"threshold_uncertainty_score":0.06228751},"labels":[],"label_agreement":null},{"id":"W2751665091","doi":"10.1177/0265532217725776","title":"Developing and evaluating a computerized adaptive testing version of the Word Part Levels Test","year":2017,"lang":"en","type":"article","venue":"Language Testing","topic":"Second Language Acquisition and Learning","field":"Psychology","cited_by":45,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Affix; Test (biology); Vocabulary; Computerized adaptive testing; Natural language processing; Computer science; Strengths and weaknesses; Word (group theory); Vocabulary development; Psychology; Artificial intelligence; Linguistics; Psychometrics; Social psychology","score_opus":0.1617082178408199,"score_gpt":0.3865890369439115,"score_spread":0.2248808191030916,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2751665091","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9786598,0.00021194761,0.010724606,0.00015273923,0.00008876573,0.0060715494,0.00070795155,0.00025293708,0.003129761],"genre_scores_gemma":[0.8737003,0.000488284,0.10972172,0.0002997024,0.00006095397,0.009254976,0.0033691775,0.0001129712,0.002991911],"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98837775,0.0049148183,0.0017676428,0.0011629924,0.0033865625,0.00039029904],"domain_scores_gemma":[0.9744304,0.012791255,0.0015619867,0.0014082099,0.008850729,0.00095738686],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016080242,0.0007735039,0.00070386275,0.0016096008,0.00051029975,0.0012565461,0.0013499465,0.0009056579,0.0013992158],"category_scores_gemma":[0.037252482,0.00042355174,0.00078139134,0.0010900326,0.0007604887,0.0014981584,0.0012821344,0.0011836517,0.0006376651],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002838269,0.012719089,0.53953636,0.00048418704,0.0002866397,0.0005901551,0.005836051,0.005216554,0.016838005,0.0010230151,0.0033665383,0.41126513],"study_design_scores_gemma":[0.0013382739,0.026635507,0.9034678,0.00019281199,0.0004755202,0.0019035868,0.0026006184,0.021101777,0.029013216,0.00092435797,0.012123493,0.00022317107],"about_ca_topic_score_codex":0.007416,"about_ca_topic_score_gemma":0.009491154,"teacher_disagreement_score":0.016080242,"about_ca_system_score_codex":0.0013363513,"about_ca_system_score_gemma":0.0032395213,"threshold_uncertainty_score":0.08504146},"labels":[],"label_agreement":null},{"id":"W2776964473","doi":"10.1177/0265532217716732","title":"The development of EFL examinations in Haiti: Collaboration and language assessment literacy development","year":2017,"lang":"en","type":"article","venue":"Language Testing","topic":"Multilingual Education and Policy","field":"Social Sciences","cited_by":87,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; University of Ottawa","funders":"","keywords":"Christian ministry; Psychology; Literacy; Medical education; Professional development; English language; Pedagogy; Faculty development; Mathematics education; Language assessment; Language development; Political science; Medicine","score_opus":0.07883816509620944,"score_gpt":0.5035994185271453,"score_spread":0.42476125343093585,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2776964473","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9862373,0.00016604995,0.0019053932,0.0030149068,0.000020898868,0.00061326975,0.0000338934,0.000026563457,0.007981763],"genre_scores_gemma":[0.99323624,0.0001463185,0.0046864133,0.00029850172,0.000004999048,0.0003460521,0.00002697889,0.000007432537,0.0012471784],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.96617323,0.027540509,0.0009018516,0.0010466975,0.0013802758,0.0029574172],"domain_scores_gemma":[0.9612841,0.019464804,0.0036788208,0.0016714571,0.006207282,0.0076935505],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.043172657,0.0002214107,0.00035978414,0.001689405,0.010843241,0.004123246,0.0019117328,0.0014044318,0.0017804836],"category_scores_gemma":[0.049670145,0.0007273566,0.00017779226,0.001272293,0.0035395205,0.003543017,0.010417621,0.0019340107,0.00041721194],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018231446,0.0014667796,0.101526015,0.00023616813,0.000017424552,0.0024267503,0.76362085,0.00037205894,0.0019654327,0.0027371265,0.0013736951,0.12407535],"study_design_scores_gemma":[0.00007261232,0.00079671195,0.08785702,0.00040994262,0.000019827114,0.0006328819,0.8857672,0.0011888606,0.0021871561,0.0017188204,0.019280933,0.00006793575],"about_ca_topic_score_codex":0.03409944,"about_ca_topic_score_gemma":0.06795359,"teacher_disagreement_score":0.043172657,"about_ca_system_score_codex":0.009886752,"about_ca_system_score_gemma":0.04395098,"threshold_uncertainty_score":0.22832155},"labels":[],"label_agreement":null},{"id":"W2782384206","doi":"10.1177/0265532217750692","title":"Examining sources of variability in repeaters’ L2 writing scores: The case of the PTE Academic writing section","year":2018,"lang":"en","type":"article","venue":"Language Testing","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Test (biology); Psychology; Context (archaeology); Boston Naming Test; Language assessment; Language proficiency; Mathematics education; Cognition","score_opus":0.3740090240133829,"score_gpt":0.4424069973241175,"score_spread":0.0683979733107346,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2782384206","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.991251,0.00037361067,0.006684916,0.00016397385,0.00002614572,0.000063297004,0.0003714664,0.000054415606,0.0010111765],"genre_scores_gemma":[0.9973416,0.00005012375,0.001704032,0.000031216237,0.000016863425,0.000044172757,0.00037149104,0.000029764906,0.0004107461],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.97866493,0.009484821,0.0015199538,0.0040857247,0.005554582,0.0006899992],"domain_scores_gemma":[0.88283396,0.06778924,0.020327954,0.018474314,0.009355581,0.0012189422],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01680917,0.00059038145,0.00094039575,0.0020775185,0.0007889246,0.001449179,0.0012321948,0.00091543584,0.00087500195],"category_scores_gemma":[0.086440854,0.00042988634,0.0012075158,0.0018309576,0.00088221283,0.0011721579,0.0020513262,0.0012586119,0.00030545142],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016506332,0.000081511374,0.9730865,0.00004991613,0.0004904555,0.00022618138,0.006504957,0.0004934868,0.0011006834,0.00023356844,0.00025762478,0.017309906],"study_design_scores_gemma":[0.0000043153273,0.00016482918,0.99447167,0.00002485887,0.000065223445,0.00046772,0.0013166744,0.0019209227,0.000724516,0.00025486283,0.00055417465,0.000030214878],"about_ca_topic_score_codex":0.009939506,"about_ca_topic_score_gemma":0.013003569,"teacher_disagreement_score":0.01680917,"about_ca_system_score_codex":0.00078676915,"about_ca_system_score_gemma":0.0007109115,"threshold_uncertainty_score":0.08889645},"labels":[],"label_agreement":null},{"id":"W2917835789","doi":"10.1177/0265532219828252","title":"The Test of English for International Communication (TOEIC <sup>®</sup> )","year":2019,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":22,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"American Psychological Association; Educational Testing Service","keywords":"TOEIC; Psychology; Test (biology); Linguistics; International communication; Language proficiency; Language assessment; Mathematics education; Communication; Philosophy; Reading (process)","score_opus":0.02662448091769125,"score_gpt":0.2592560255326446,"score_spread":0.23263154461495333,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2917835789","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.42523396,0.0023612843,0.030000614,0.003756622,0.0033633348,0.002579929,0.039510977,0.0033765342,0.48981673],"genre_scores_gemma":[0.6584084,0.0016576409,0.044694003,0.0031153162,0.00041264918,0.0046030064,0.032655943,0.0013040277,0.25314903],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99802905,0.00042167233,0.00024501068,0.00015338942,0.0009356521,0.00021525273],"domain_scores_gemma":[0.995529,0.0013741769,0.00041668938,0.00029715386,0.0017646288,0.00061843893],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016131512,0.0011229026,0.00054048584,0.0022664883,0.000584983,0.0011503071,0.00080420467,0.0010963571,0.026687011],"category_scores_gemma":[0.010815826,0.0001733513,0.0006431573,0.000704015,0.00066266936,0.0015752165,0.0016317655,0.00084618386,0.014806698],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001877107,0.001144678,0.1533696,0.0005434066,0.000116422525,0.0020491597,0.0016748648,0.0008399543,0.027505796,0.0062048063,0.1947449,0.6099294],"study_design_scores_gemma":[0.0003174195,0.0032197167,0.65307623,0.0005114068,0.00012089815,0.010580656,0.0027960436,0.0030569155,0.052797977,0.005450765,0.26782504,0.00024697775],"about_ca_topic_score_codex":0.0035250543,"about_ca_topic_score_gemma":0.0034498281,"teacher_disagreement_score":0.026687011,"about_ca_system_score_codex":0.00032774237,"about_ca_system_score_gemma":0.0013254061,"threshold_uncertainty_score":0.08927691},"labels":[],"label_agreement":null},{"id":"W2950885273","doi":"10.1177/0265532219854982","title":"Book Review: Assessment in the Language Classroom: Teachers Supporting Student Learning","year":2019,"lang":"en","type":"article","venue":"Language Testing","topic":"Second Language Learning and Teaching","field":"Arts and Humanities","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Psychology; Mathematics education; Language assessment; Pedagogy","score_opus":0.018881440790615857,"score_gpt":0.31109286189690133,"score_spread":0.29221142110628545,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2950885273","genre_codex":"review","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00025977712,0.8469016,0.0007469294,0.047337286,0.08965211,0.00020767604,0.0003340494,0.00010934798,0.014451193],"genre_scores_gemma":[0.0027861346,0.79016304,0.0012873727,0.052178178,0.07329344,0.00040267862,0.0006857414,0.00014553181,0.0790579],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99744785,0.0006316674,0.00029079043,0.00027595385,0.0012023136,0.00015129232],"domain_scores_gemma":[0.98423773,0.0069759954,0.0010277272,0.00023356198,0.006518669,0.0010063206],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0022603483,0.0012022193,0.0033093134,0.0037474853,0.0008326493,0.0027915784,0.0026537233,0.0064367573,0.02080447],"category_scores_gemma":[0.012706642,0.0007749061,0.0011006717,0.0047268164,0.001357651,0.0020175616,0.0012631153,0.0050664577,0.012535314],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000038416365,0.000035189052,0.00005052771,0.0022308824,0.000025089194,0.000041219264,0.000012879363,0.00006555737,0.00012013906,0.00026359118,0.941859,0.055257384],"study_design_scores_gemma":[0.000075340795,0.00010485901,0.0010021168,0.0045567276,0.000083223844,0.00036139757,0.000034797766,0.00006529481,0.00009595179,0.00047733786,0.99311876,0.000024064066],"about_ca_topic_score_codex":0.011494175,"about_ca_topic_score_gemma":0.040122714,"teacher_disagreement_score":0.02080447,"about_ca_system_score_codex":0.0029307338,"about_ca_system_score_gemma":0.006360134,"threshold_uncertainty_score":0.06959784},"labels":[],"label_agreement":null},{"id":"W3039959558","doi":"10.1177/0265532220930348","title":"Change in home language environment and English literacy achievement over time: A multi-group latent growth curve modeling investigation","year":2020,"lang":"en","type":"article","venue":"Language Testing","topic":"Multilingual Education and Policy","field":"Social Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Latent growth modeling; Literacy; Psychology; Longitudinal study; Home language; Competence (human resources); Population; Academic achievement; Mathematics education; Immigration; Developmental psychology; Pedagogy; Social psychology; Demography; Sociology; Geography; Medicine","score_opus":0.08900675997372587,"score_gpt":0.3676722898153619,"score_spread":0.278665529841636,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3039959558","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9857897,0.00019598297,0.011740765,0.00047960435,0.000023704568,0.00015552783,0.0008721963,0.0001054877,0.00063707534],"genre_scores_gemma":[0.99163824,0.00011180121,0.004944407,0.000035486584,0.000008064278,0.000224932,0.0013381441,0.000034986595,0.0016639843],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9943381,0.003421572,0.00019176658,0.00096341316,0.00047501965,0.000610163],"domain_scores_gemma":[0.98695403,0.0075101177,0.0017092134,0.0018422282,0.0012831398,0.0007012852],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.014800913,0.0012288357,0.0013546932,0.002320973,0.001674204,0.0024509614,0.0025854746,0.0013382827,0.003474491],"category_scores_gemma":[0.019586358,0.00054868107,0.0033734064,0.002736117,0.0013965593,0.001483276,0.0029909515,0.002644012,0.00084327126],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00054430455,0.00060001755,0.95848894,0.000073111696,0.0006939907,0.00025514935,0.0052527664,0.012920024,0.00047079544,0.0026074685,0.0012996638,0.016793631],"study_design_scores_gemma":[0.00010137774,0.0007735501,0.5940625,0.00017032554,0.0005384912,0.000351403,0.0108237015,0.38358733,0.0005651512,0.0047159805,0.004175321,0.00013487061],"about_ca_topic_score_codex":0.18675473,"about_ca_topic_score_gemma":0.10689347,"teacher_disagreement_score":0.18675473,"about_ca_system_score_codex":0.0030074422,"about_ca_system_score_gemma":0.004235401,"threshold_uncertainty_score":0.3713354},"labels":[],"label_agreement":null},{"id":"W3040446934","doi":"10.1177/0265532220937830","title":"More efficient processes for creating automated essay scoring frameworks: A demonstration of two algorithms","year":2020,"lang":"en","type":"article","venue":"Language Testing","topic":"Topic Modeling","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Rubric; Artificial intelligence; Machine learning; Computer science; Support vector machine; Convolutional neural network; Feature engineering; Deep learning; Artificial neural network; Natural language processing; Meaning (existential); Feature (linguistics); Strengths and weaknesses; F1 score; Algorithm; Mathematics; Mathematics education; Linguistics; Psychology","score_opus":0.0348837319221498,"score_gpt":0.30971148316334696,"score_spread":0.2748277512411972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3040446934","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.017618459,0.00014805316,0.9709029,0.00033947593,0.00013898258,0.00035376666,0.00018299396,0.0073189735,0.0029963304],"genre_scores_gemma":[0.1304706,0.000119029675,0.864866,0.0000875427,0.00008666054,0.00035599706,0.00045089802,0.00045328803,0.0031100363],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99120927,0.0032655546,0.00083529373,0.0014749842,0.0028805584,0.00033440947],"domain_scores_gemma":[0.9843061,0.005490905,0.0009894988,0.003672051,0.0048962263,0.0006451946],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008945438,0.0013842066,0.0010488599,0.0028678663,0.0006793856,0.0032235489,0.0022656021,0.0014739535,0.005215978],"category_scores_gemma":[0.03084167,0.0006046362,0.0009189309,0.0013767404,0.0008801701,0.0040600374,0.0039327038,0.002276634,0.0033694266],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036456683,0.00046583585,0.0055950866,0.00019982447,0.00009232007,0.00017311797,0.00080955884,0.025253091,0.018724134,0.027876733,0.00933248,0.91111314],"study_design_scores_gemma":[0.00009948329,0.00022080555,0.0040405095,0.00006183771,0.000032698623,0.0003489883,0.00030757955,0.9240695,0.031339303,0.02094692,0.018422658,0.00010967359],"about_ca_topic_score_codex":0.0040064002,"about_ca_topic_score_gemma":0.003722943,"teacher_disagreement_score":0.008945438,"about_ca_system_score_codex":0.0011260927,"about_ca_system_score_gemma":0.001913853,"threshold_uncertainty_score":0.047308564},"labels":[],"label_agreement":null},{"id":"W3087153552","doi":"10.1177/0265532220957298","title":"Hanyu Shuiping Kaoshi (HSK): A multi-level, multi-purpose proficiency test","year":2020,"lang":"en","type":"article","venue":"Language Testing","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":50,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Language proficiency; Test (biology); Psychology; Argument (complex analysis); Language assessment; Scale (ratio); Mathematics education; Linguistics; Geography","score_opus":0.8227491867455936,"score_gpt":0.5099737195301289,"score_spread":0.3127754672154647,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3087153552","genre_codex":"review","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.2014338,0.66051763,0.042494725,0.0206019,0.0039051247,0.0055872086,0.006327467,0.0013643621,0.057767794],"genre_scores_gemma":[0.65713537,0.24742843,0.067030996,0.0033936154,0.0012200105,0.0035457432,0.0068217535,0.00013952676,0.013284533],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9964742,0.001218317,0.0006682827,0.00023123992,0.0013322646,0.00007565485],"domain_scores_gemma":[0.98776346,0.005560908,0.0013616211,0.0004230697,0.004363662,0.0005271061],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007883831,0.0003635658,0.000728913,0.0041079763,0.0002901108,0.0008506479,0.0007496743,0.0005461299,0.002374853],"category_scores_gemma":[0.018312225,0.00015534232,0.000503607,0.002219226,0.0005447093,0.0010553906,0.00087291974,0.00060514524,0.0005317453],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016638539,0.00019277606,0.016185068,0.0065378114,0.0002786633,0.00018776221,0.00017597195,0.00020118547,0.0014587537,0.0010564033,0.0088497,0.9647096],"study_design_scores_gemma":[0.0008867993,0.0052269967,0.4557226,0.018183274,0.0042953826,0.005888433,0.0010759794,0.0047769574,0.021785386,0.005644336,0.47618687,0.00032699274],"about_ca_topic_score_codex":0.0047076014,"about_ca_topic_score_gemma":0.00616891,"teacher_disagreement_score":0.007883831,"about_ca_system_score_codex":0.0009432597,"about_ca_system_score_gemma":0.008148988,"threshold_uncertainty_score":0.041694164},"labels":[],"label_agreement":null},{"id":"W3214027731","doi":"10.1177/02655322211052680","title":"Investigating and optimizing score dependability of a local ITA speaking test across language groups: A generalizability theory approach","year":2021,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Generalizability theory; Dependability; Language proficiency; Psychology; Variance (accounting); Test (biology); Construct (python library); Formative assessment; Computer science; Mathematics education; Developmental psychology; Accounting","score_opus":0.050868985531210546,"score_gpt":0.3497865942429446,"score_spread":0.29891760871173406,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3214027731","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6361603,0.00041987657,0.35266072,0.00062853436,0.000056910125,0.0011100963,0.00020248465,0.00043369824,0.0083274385],"genre_scores_gemma":[0.9507976,0.00006967164,0.047814503,0.000080312566,0.000025289792,0.00056433893,0.00015736393,0.0000704249,0.00042043568],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.94186175,0.042277265,0.0023312175,0.005780945,0.006960205,0.00078858994],"domain_scores_gemma":[0.73534185,0.21294822,0.010937085,0.025207106,0.014718775,0.0008469287],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.08313917,0.0015327206,0.0014003136,0.003990026,0.0008686305,0.0025761104,0.0017621223,0.0010370555,0.0016551898],"category_scores_gemma":[0.24767268,0.0006271307,0.0026763245,0.0029654484,0.0028935762,0.0032575608,0.0035016646,0.0017916902,0.00029129005],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00093442044,0.0006063585,0.6727764,0.00038435456,0.002189075,0.00019474018,0.008351212,0.0154744,0.006535386,0.009704689,0.0005476653,0.28230137],"study_design_scores_gemma":[0.00023132478,0.005896743,0.8465566,0.00021669985,0.001621534,0.00038176344,0.004894767,0.097275615,0.01561534,0.02442326,0.0027327163,0.00015356738],"about_ca_topic_score_codex":0.0043842136,"about_ca_topic_score_gemma":0.0036465917,"teacher_disagreement_score":0.08313917,"about_ca_system_score_codex":0.0018039201,"about_ca_system_score_gemma":0.0019270694,"threshold_uncertainty_score":0.43968725},"labels":[],"label_agreement":null},{"id":"W4251721468","doi":"10.1177/0265532220925448","title":"Understanding writing quality change: A longitudinal study of repeaters of a high-stakes standardized English proficiency test","year":2020,"lang":"en","type":"article","venue":"Language Testing","topic":"Writing and Handwriting Education","field":"Social Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Language proficiency; Psychology; Test (biology); Sophistication; Linguistics; Mathematics education","score_opus":0.4041793967710166,"score_gpt":0.4111835522633015,"score_spread":0.007004155492284891,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4251721468","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99967337,0.00004487671,0.00005598756,0.000020005553,0.0000020465513,0.000007708416,0.000060598326,0.000002364795,0.0001330857],"genre_scores_gemma":[0.99930084,0.000025698253,0.00007934061,0.000012171919,0.000002639247,0.000009019112,0.00016590845,0.0000026831442,0.0004017013],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99891376,0.00021864811,0.000099599696,0.00018015013,0.0004119833,0.00017588248],"domain_scores_gemma":[0.9937552,0.0008144168,0.0024851032,0.00053242856,0.0016448133,0.00076802087],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017725579,0.0003104826,0.0004894471,0.0009956155,0.0008225604,0.0007769205,0.00066962134,0.0008115863,0.00081677706],"category_scores_gemma":[0.008109648,0.00032816757,0.00053059525,0.0006929956,0.00040827694,0.0007759631,0.00062730134,0.0009766462,0.00046802213],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007691726,0.0001943209,0.9940573,0.0000063444454,0.00003903639,0.00016033625,0.0016939439,0.000027741147,0.00046120878,0.000009713145,0.00006621791,0.0032067893],"study_design_scores_gemma":[0.0000019821805,0.0002879178,0.9984547,0.0000028391257,0.000009066498,0.0001885287,0.00070014887,0.00009094222,0.00011811143,0.000007958578,0.00013230552,0.000005528087],"about_ca_topic_score_codex":0.018969724,"about_ca_topic_score_gemma":0.020837972,"teacher_disagreement_score":0.018969724,"about_ca_system_score_codex":0.0005514867,"about_ca_system_score_gemma":0.0004344385,"threshold_uncertainty_score":0.037718594},"labels":[],"label_agreement":null},{"id":"W4251941390","doi":"10.1177/0265532220929918","title":"Automated scoring of junior and senior high essays using Coh-Metrix features: Implications for large-scale language testing","year":2020,"lang":"en","type":"article","venue":"Language Testing","topic":"Software Engineering Research","field":"Computer Science","cited_by":58,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Centre for Advancing Health Outcomes; University of Alberta","funders":"","keywords":"Natural language processing; Artificial intelligence; Rating scale; Computer science; Scale (ratio); Quality (philosophy); Construct (python library); Psychology; Computational linguistics; Disadvantaged; Machine learning; Developmental psychology","score_opus":0.0384669575825602,"score_gpt":0.314934657209404,"score_spread":0.2764676996268438,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4251941390","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9057022,0.00015473893,0.08502079,0.000737787,0.000078240904,0.0006546102,0.0009435304,0.0016023803,0.005105739],"genre_scores_gemma":[0.9451553,0.000028603572,0.05261295,0.00005831062,0.000029436469,0.00038280082,0.00070253544,0.000082795756,0.000947226],"study_design_codex":"observational","study_design_gemma":"not_applicable","domain_scores_codex":[0.9826245,0.011149075,0.001145939,0.0013772713,0.003304211,0.00039899288],"domain_scores_gemma":[0.87404466,0.0725709,0.013568377,0.014366395,0.02334865,0.0021009862],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.022005778,0.000703687,0.0005234812,0.0018903319,0.00061810634,0.0018867017,0.0011379193,0.0005495879,0.0016732344],"category_scores_gemma":[0.1303999,0.00025200358,0.0004401214,0.0019251253,0.00070892076,0.0020835954,0.0017141878,0.0012150668,0.00063632644],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000787052,0.00081563153,0.51557225,0.00023529692,0.0001439789,0.00012137948,0.0030621337,0.010712434,0.00787936,0.002254873,0.006686268,0.4517294],"study_design_scores_gemma":[0.00013234996,0.0017263475,0.665396,0.00016275565,0.00006515012,0.00031676202,0.0032229463,0.29533166,0.018650383,0.0063330657,0.008492175,0.00017050953],"about_ca_topic_score_codex":0.0041532195,"about_ca_topic_score_gemma":0.009832878,"teacher_disagreement_score":0.022005778,"about_ca_system_score_codex":0.0011107483,"about_ca_system_score_gemma":0.001674405,"threshold_uncertainty_score":0.11637908},"labels":[],"label_agreement":null},{"id":"W4291825485","doi":"10.1177/02655322221114895","title":"Book Review: <i>Multilingual Testing and Assessment</i>","year":2022,"lang":"en","type":"article","venue":"Language Testing","topic":"Multilingual Education and Policy","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Psychology; Linguistics; Language assessment; Mathematics education; Philosophy","score_opus":0.0868325680279467,"score_gpt":0.47609491008186594,"score_spread":0.38926234205391924,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4291825485","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00015811734,0.4369724,0.0010157306,0.21967745,0.24330369,0.00015720617,0.000836432,0.00030027973,0.09757873],"genre_scores_gemma":[0.0018619087,0.2841379,0.0011158532,0.1602399,0.13820256,0.00029551983,0.000979739,0.00030976196,0.4128569],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99782145,0.0003470077,0.00013530889,0.00019361079,0.0013610293,0.00014160627],"domain_scores_gemma":[0.9868672,0.0053334087,0.00069776695,0.0002854486,0.0057106414,0.0011055428],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020475637,0.0011182656,0.0019958054,0.0046276357,0.0011618276,0.004021074,0.002334514,0.0065075797,0.061115157],"category_scores_gemma":[0.013840329,0.00066503143,0.000984173,0.0046202308,0.0015491887,0.0031775879,0.0018625505,0.006834511,0.040555153],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000025118568,0.0000031500392,0.00001074608,0.00009781996,0.0000014517726,0.000007969869,0.0000032652802,0.000010190634,0.000009332528,0.00022630084,0.9873351,0.012292182],"study_design_scores_gemma":[0.0000058871788,0.000007084774,0.00017437937,0.00055600127,0.0000058681317,0.00009962423,0.000014512052,0.000017310791,0.000019982122,0.00043032866,0.9986625,0.0000064843257],"about_ca_topic_score_codex":0.018132912,"about_ca_topic_score_gemma":0.059334747,"teacher_disagreement_score":0.061115157,"about_ca_system_score_codex":0.0042763543,"about_ca_system_score_gemma":0.008087053,"threshold_uncertainty_score":0.2044506},"labels":[],"label_agreement":null},{"id":"W4324373876","doi":"10.1177/02655322231156819","title":"Ukrainian language proficiency test review","year":2023,"lang":"en","type":"article","venue":"Language Testing","topic":"Linguistics, Language Diversity, and Identity","field":"Arts and Humanities","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Western University","funders":"","keywords":"Ukrainian; Language proficiency; Psychology; Language assessment; Test (biology); Linguistics; Mathematics education","score_opus":0.06068389481600776,"score_gpt":0.28133156771583895,"score_spread":0.2206476728998312,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4324373876","genre_codex":"review","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0023179983,0.9915446,0.00017094536,0.0012700455,0.00034598564,0.000024348388,0.0005739097,0.000011572064,0.003740533],"genre_scores_gemma":[0.024568386,0.97035193,0.0005060087,0.001192781,0.0002845,0.00006887336,0.0010685669,0.000012575014,0.0019464588],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99892765,0.0002456725,0.00029312578,0.00014521091,0.00032404164,0.00006429123],"domain_scores_gemma":[0.99640936,0.0017954739,0.00039645287,0.00007129971,0.001227213,0.00010006639],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0024718554,0.00042149355,0.0011467268,0.004310525,0.0003397088,0.001186835,0.0011788461,0.0005777032,0.0039014327],"category_scores_gemma":[0.0067140004,0.0001676235,0.0006251024,0.0036539934,0.0005934352,0.0011486587,0.00085862685,0.00046353403,0.00076293777],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016778271,0.00007402451,0.010731305,0.020114489,0.00037355552,0.00042198633,0.00021598447,0.00017582672,0.00022822821,0.0022097612,0.038247578,0.9270395],"study_design_scores_gemma":[0.000051629355,0.00022237968,0.076528564,0.055704184,0.0019121219,0.0033280961,0.00088742666,0.000120323,0.00074957573,0.0013715558,0.85908365,0.000040505758],"about_ca_topic_score_codex":0.015050163,"about_ca_topic_score_gemma":0.020859463,"teacher_disagreement_score":0.015050163,"about_ca_system_score_codex":0.0017434782,"about_ca_system_score_gemma":0.008802568,"threshold_uncertainty_score":0.029925168},"labels":[],"label_agreement":null},{"id":"W4375945263","doi":"10.1177/02655322231164565","title":"Book review: Learning-Oriented Language Assessment: Putting Theory into Practice","year":2023,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Psychology; Linguistics; Language assessment; Mathematics education; Philosophy","score_opus":0.019022381633885147,"score_gpt":0.40005753558139345,"score_spread":0.3810351539475083,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4375945263","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00015551728,0.8510592,0.0010249078,0.0672956,0.07443466,0.00012632896,0.00022171886,0.000102347534,0.005579729],"genre_scores_gemma":[0.0029593967,0.7681321,0.0025471991,0.094549134,0.09060175,0.0004261216,0.00053317804,0.00016327458,0.040087845],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9947179,0.0015750612,0.0005590842,0.0004641778,0.0024857074,0.0001980981],"domain_scores_gemma":[0.9607786,0.021161446,0.0024328984,0.0004800741,0.013557679,0.0015892391],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045517557,0.0011751009,0.004131925,0.005479272,0.00073048007,0.0033206062,0.0029765435,0.0067787697,0.0150992675],"category_scores_gemma":[0.032985445,0.00082830363,0.0013546041,0.0054009156,0.0017339397,0.0025798487,0.0015307604,0.006316013,0.010567267],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003003021,0.000023892017,0.000057545254,0.0028597182,0.00004407664,0.000037998,0.0000129405535,0.00007197631,0.00006313363,0.000330939,0.9384242,0.058043543],"study_design_scores_gemma":[0.00013697085,0.00010838252,0.0011900619,0.01062917,0.00017664224,0.0006301167,0.000051709485,0.00018521021,0.00012020454,0.001238914,0.9854903,0.000042285636],"about_ca_topic_score_codex":0.009466861,"about_ca_topic_score_gemma":0.030305713,"teacher_disagreement_score":0.0150992675,"about_ca_system_score_codex":0.0042643473,"about_ca_system_score_gemma":0.0073432853,"threshold_uncertainty_score":0.050512075},"labels":[],"label_agreement":null},{"id":"W4381740816","doi":"10.1177/02655322231179128","title":"Rethinking student placement to enhance efficiency and student agency","year":2023,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Université de Montréal; Carleton University; University of Ottawa","funders":"University of Ottawa","keywords":"Agency (philosophy); Context (archaeology); Psychology; Process (computing); Test (biology); Mathematics education; Language proficiency; Pedagogy; Computer science; Sociology","score_opus":0.0478632348052326,"score_gpt":0.4160427232575863,"score_spread":0.3681794884523537,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4381740816","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8548789,0.00019039275,0.108494416,0.003935029,0.0006260409,0.003991201,0.00010342672,0.0023268561,0.025453698],"genre_scores_gemma":[0.807312,0.00022320547,0.18212306,0.0004379776,0.00009896883,0.0018284615,0.00011595993,0.00032970126,0.0075307144],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.93603384,0.04257496,0.003972036,0.0022478437,0.012835615,0.0023356918],"domain_scores_gemma":[0.8475259,0.08954683,0.0076497067,0.021708902,0.025743144,0.007825507],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.060859933,0.0011278093,0.0011995267,0.002030472,0.0022817967,0.009363875,0.0029899322,0.0012839333,0.0042055994],"category_scores_gemma":[0.18690352,0.000583939,0.00085714326,0.0012219523,0.0027535472,0.004686075,0.005997197,0.0030837492,0.002235246],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00041952622,0.0041967407,0.038171753,0.00052994426,0.00005951537,0.00023481432,0.058445252,0.0016487245,0.014439194,0.0040545855,0.0074749943,0.870325],"study_design_scores_gemma":[0.0008399876,0.02330497,0.37568027,0.0024289961,0.00026038443,0.0015605697,0.19154115,0.043473963,0.10691099,0.048011284,0.20459655,0.001390902],"about_ca_topic_score_codex":0.001347309,"about_ca_topic_score_gemma":0.003621288,"teacher_disagreement_score":0.060859933,"about_ca_system_score_codex":0.0028215016,"about_ca_system_score_gemma":0.0056279376,"threshold_uncertainty_score":0.32186192},"labels":[],"label_agreement":null},{"id":"W4383069305","doi":"10.1177/02655322231179134","title":"Fairness of using different English accents: The effect of shared L1s in listening tasks of the Duolingo English test","year":2023,"lang":"en","type":"article","venue":"Language Testing","topic":"Phonetics and Phonology Research","field":"Psychology","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Brock University","funders":"","keywords":"Active listening; Psychology; Stress (linguistics); Test (biology); Interlanguage; Vocabulary; Dictation; Linguistics; Task (project management); Hindi; Communication","score_opus":0.038221913593031565,"score_gpt":0.34834213137898773,"score_spread":0.31012021778595616,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4383069305","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9928139,0.0001444125,0.004023078,0.00012740177,0.000038644557,0.00008621126,0.000022057802,0.000021647222,0.002722617],"genre_scores_gemma":[0.9971232,0.000032987722,0.002153491,0.00014898278,0.000027961793,0.000049529273,0.000024166857,0.00002279969,0.00041692657],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.95171,0.03043661,0.0037531361,0.004363387,0.0088155875,0.0009212358],"domain_scores_gemma":[0.69814545,0.24035902,0.02795635,0.021171587,0.0088105155,0.0035571235],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05847613,0.0005535468,0.00066006614,0.0006298637,0.0010852917,0.0019147515,0.0007575943,0.00076788815,0.0011827084],"category_scores_gemma":[0.16406351,0.000420373,0.00062028307,0.00031713728,0.0023981722,0.0018629762,0.0036947534,0.0012667967,0.00033765205],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.017618418,0.002189803,0.7263381,0.00038652608,0.0008199843,0.000557832,0.029897574,0.0017403275,0.0844661,0.0015476902,0.00042346763,0.13401425],"study_design_scores_gemma":[0.00022342117,0.005909177,0.9531065,0.00012053517,0.00029404255,0.00071062846,0.004002206,0.002249994,0.030413255,0.0014869077,0.0013368108,0.00014634863],"about_ca_topic_score_codex":0.0010908948,"about_ca_topic_score_gemma":0.0022642564,"teacher_disagreement_score":0.05847613,"about_ca_system_score_codex":0.00063399225,"about_ca_system_score_gemma":0.00061809516,"threshold_uncertainty_score":0.30925506},"labels":[],"label_agreement":null},{"id":"W4387429753","doi":"10.1177/02655322231202947","title":"Our validity looks like justice. Does yours?","year":2023,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Lethbridge","funders":"","keywords":"Economic Justice; Oppression; Psychology; Licensure; Set (abstract data type); Social psychology; Mathematics education; Pedagogy; Law; Political science; Computer science","score_opus":0.13450710571791363,"score_gpt":0.4132933080995925,"score_spread":0.2787862023816789,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387429753","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.055092845,0.004960268,0.07209195,0.6691323,0.010543173,0.0005949302,0.00052160094,0.00040721486,0.1866557],"genre_scores_gemma":[0.87075263,0.0016521087,0.0331813,0.078876786,0.003154908,0.0006152446,0.0001900357,0.00041892048,0.011157936],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.8181638,0.10470825,0.01075343,0.013887708,0.048504703,0.0039820834],"domain_scores_gemma":[0.56779134,0.20396556,0.03484373,0.06526697,0.11840912,0.009723244],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.14140457,0.00080382277,0.0016495985,0.0048978818,0.009235154,0.013172378,0.0020948888,0.004590561,0.004955603],"category_scores_gemma":[0.41526487,0.0006694653,0.0013567093,0.0025301906,0.06089355,0.020363461,0.009169753,0.009881493,0.0018844545],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002285718,0.00032415052,0.04270772,0.0011424083,0.00044750332,0.0002711135,0.058484796,0.00047207205,0.0017390798,0.63451684,0.05859088,0.20107497],"study_design_scores_gemma":[0.000073776784,0.00036838718,0.018010205,0.0026457056,0.00021600362,0.0004913288,0.033465,0.0014202632,0.0021137686,0.7494276,0.19152565,0.0002422938],"about_ca_topic_score_codex":0.009504145,"about_ca_topic_score_gemma":0.007370447,"teacher_disagreement_score":0.85859543,"about_ca_system_score_codex":0.0069159847,"about_ca_system_score_gemma":0.015258612,"threshold_uncertainty_score":0.74782777},"labels":[],"label_agreement":null},{"id":"W4399235559","doi":"10.1177/02655322241249754","title":"A scoping review of research on second language test preparation","year":2024,"lang":"en","type":"review","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Sherbrooke; Western University","funders":"","keywords":"Psychology; Test (biology); Language assessment; Language proficiency; Linguistics; Mathematics education","score_opus":0.2975427459650408,"score_gpt":0.5064660563792833,"score_spread":0.2089233104142425,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399235559","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0003034035,0.99721044,0.0003312869,0.0005368813,0.00026372424,0.00015566633,0.00015029535,0.000010227523,0.0010381977],"genre_scores_gemma":[0.0017971856,0.99670655,0.0006047163,0.00026326263,0.00006814412,0.0002355154,0.0001535798,0.000005350446,0.00016569914],"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","domain_scores_codex":[0.98967105,0.0029809584,0.0039500566,0.00071746914,0.0024136729,0.00026684086],"domain_scores_gemma":[0.93602437,0.049088266,0.005240827,0.0011412802,0.0080217,0.00048361864],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015138312,0.0016426281,0.0033796222,0.020579956,0.0014461227,0.0038756246,0.0021235945,0.0025913948,0.0061447453],"category_scores_gemma":[0.071209595,0.001115859,0.004001724,0.023527894,0.0015571794,0.003956672,0.0024065753,0.00214661,0.001254846],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006536722,0.000039654635,0.00057372137,0.61834884,0.00063993357,0.00019693107,0.0011566622,0.00018992247,0.00035531865,0.0024181078,0.010969366,0.3650461],"study_design_scores_gemma":[0.000013811949,0.00005164253,0.0013537972,0.88312274,0.002077254,0.00034704548,0.0006666887,0.000043798776,0.00019783922,0.0009581991,0.11114564,0.000021554597],"about_ca_topic_score_codex":0.009294444,"about_ca_topic_score_gemma":0.023329645,"teacher_disagreement_score":0.020579956,"about_ca_system_score_codex":0.003960383,"about_ca_system_score_gemma":0.025978189,"threshold_uncertainty_score":0.080060005},"labels":[],"label_agreement":null},{"id":"W4404521715","doi":"10.1177/02655322241291764","title":"Review of the Canadian English Language Proficiency Index Program (CELPIP)","year":2024,"lang":"en","type":"article","venue":"Language Testing","topic":"Multilingual Education and Policy","field":"Social Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Ottawa","funders":"","keywords":"Psychology; Language proficiency; Index (typography); Language assessment; Linguistics; Mathematics education; Computer science; Philosophy","score_opus":0.06094631709999317,"score_gpt":0.45701666441974953,"score_spread":0.3960703473197564,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404521715","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00020926565,0.9937338,0.00012732855,0.0023667302,0.0005809011,0.00005752441,0.0007821648,0.00001125667,0.0021309513],"genre_scores_gemma":[0.0026314936,0.99423176,0.00050831423,0.0012560824,0.00017816879,0.00009108759,0.00065872783,0.00001029759,0.00043398386],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9932414,0.0011609172,0.0012444031,0.0005156099,0.0035044544,0.0003331156],"domain_scores_gemma":[0.96397865,0.009873432,0.002005806,0.0004564847,0.022507615,0.0011778959],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01124023,0.001329065,0.0026448194,0.0141305365,0.001641176,0.0028825512,0.004203777,0.0014849247,0.0041846237],"category_scores_gemma":[0.0482739,0.0007624007,0.0015503375,0.0218222,0.0020405313,0.0016574156,0.00143364,0.002196798,0.0009144878],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010934691,0.00004132051,0.0017170485,0.124599464,0.00041883587,0.00012882851,0.00068616273,0.00028535785,0.00017255459,0.005725892,0.19245017,0.673665],"study_design_scores_gemma":[0.000029594774,0.000054774464,0.012583615,0.17087574,0.00091044465,0.00025380045,0.00046750947,0.00008275242,0.00016711108,0.0005443992,0.81395155,0.00007876923],"about_ca_topic_score_codex":0.7186103,"about_ca_topic_score_gemma":0.76980466,"teacher_disagreement_score":0.2813897,"about_ca_system_score_codex":0.022969063,"about_ca_system_score_gemma":0.11585894,"threshold_uncertainty_score":0.5660937},"labels":[],"label_agreement":null},{"id":"W4407931412","doi":"10.1177/02655322251319284","title":"The relationship between English language proficiency test scores and academic achievement: A longitudinal study of two tests","year":2025,"lang":"en","type":"article","venue":"Language Testing","topic":"Higher Education Learning Practices","field":"Social Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"York University","funders":"","keywords":"Psychology; Language proficiency; Test (biology); Language assessment; Achievement test; Academic achievement; Mathematics education; English language; Test of English as a Foreign Language; Standardized test","score_opus":0.10730362990051857,"score_gpt":0.4529902981757598,"score_spread":0.34568666827524125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407931412","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9995515,0.00005118058,0.000052390496,0.00003801438,0.0000041009157,0.000007143208,0.00009026697,0.000001938664,0.00020357425],"genre_scores_gemma":[0.9990771,0.000040356503,0.00008264254,0.000021738002,0.0000032116193,0.00001725897,0.00023520272,0.0000020295336,0.00052050344],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9988966,0.00032353628,0.00008000054,0.00019012063,0.00024929352,0.00026037637],"domain_scores_gemma":[0.9942625,0.0007995369,0.0015642799,0.00048895925,0.0012194511,0.0016651587],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003607478,0.00032263945,0.00037874785,0.0011764981,0.0015908063,0.0012079299,0.00062492007,0.0006640399,0.00096543523],"category_scores_gemma":[0.0061850296,0.0003496382,0.0006463816,0.0008280507,0.00059091795,0.0010645842,0.0013234491,0.0019106915,0.00045680444],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000096971744,0.00036052286,0.99659926,0.0000029266764,0.000041295414,0.00005618694,0.0008394063,0.000022128479,0.00018438966,0.000029983374,0.00007802354,0.0016889045],"study_design_scores_gemma":[0.000005235174,0.00042281626,0.9977024,0.0000072038997,0.00002344721,0.00008078521,0.0011758693,0.00012742175,0.00014868853,0.000027945927,0.00026913924,0.0000089722835],"about_ca_topic_score_codex":0.032057304,"about_ca_topic_score_gemma":0.043343104,"teacher_disagreement_score":0.032057304,"about_ca_system_score_codex":0.0009516902,"about_ca_system_score_gemma":0.0014948552,"threshold_uncertainty_score":0.063741446},"labels":[],"label_agreement":null},{"id":"W4412676320","doi":"10.1177/02655322251359320","title":"Book review: Innovation in Learning-Oriented Language Assessment ChongS.ReindersH. (Eds.), Innovation in Learning-Oriented Language Assessment. Palgrave MacMillan, 2023. 333 pp. ISBN 978-3-031-18949-4 (hbk) US$169.99 ISBN 978-3-031-18952-4 (sbk) US$169.99 ISBN 978-3-031-18950-0 (ebk) US$129.99","year":2025,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Psychology; Linguistics; Sociology; Philosophy","score_opus":0.01741045735498279,"score_gpt":0.3375408814479155,"score_spread":0.3201304240929327,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412676320","genre_codex":"review","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00006622175,0.9838356,0.0002952583,0.0036199428,0.0075360187,0.000031980144,0.00019731474,0.00005194247,0.0043657036],"genre_scores_gemma":[0.00064457924,0.97168916,0.0006175718,0.0019090376,0.0055541038,0.000091289825,0.0005186516,0.000034751934,0.018940846],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9982389,0.00032029327,0.00023347029,0.00019186783,0.0009092348,0.00010611957],"domain_scores_gemma":[0.9929946,0.0032817854,0.0007981328,0.00012895395,0.002269938,0.0005265838],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00172693,0.0017989974,0.003506903,0.005709479,0.00048559517,0.0034287795,0.0022676026,0.0024246085,0.04336648],"category_scores_gemma":[0.008051196,0.0007211311,0.0012721367,0.00860929,0.0007939931,0.0030540193,0.0012655889,0.0031800687,0.031967796],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003516505,0.000023187751,0.0000764033,0.006506568,0.000043154538,0.000047672234,0.00004563451,0.00011384949,0.00012922793,0.0006548604,0.74475175,0.24757248],"study_design_scores_gemma":[0.000020032348,0.000038843056,0.0006147796,0.0060786386,0.000058008383,0.000539795,0.000045551365,0.00005722386,0.000068397,0.0007622139,0.9916985,0.000018099652],"about_ca_topic_score_codex":0.0059187147,"about_ca_topic_score_gemma":0.013530799,"teacher_disagreement_score":0.04336648,"about_ca_system_score_codex":0.0018297344,"about_ca_system_score_gemma":0.00431752,"threshold_uncertainty_score":0.14507538},"labels":[],"label_agreement":null},{"id":"W4412969591","doi":"10.1177/02655322251348956","title":"Investigating construct representativeness and linguistic equity of automated oral reading fluency assessment with prosody","year":2025,"lang":"en","type":"article","venue":"Language Testing","topic":"Reading and Literacy Development","field":"Psychology","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Institute for Christian Studies","funders":"Iran Science Elites Federation; University of Toronto","keywords":"Prosody; Fluency; Psychology; Ell; Natural language processing; Linguistics; Reading comprehension; Reading (process); Cognitive psychology; Computer science; Artificial intelligence; Mathematics education; Speech recognition; Teaching method; Vocabulary development","score_opus":0.03767559474110727,"score_gpt":0.4197115117810315,"score_spread":0.38203591703992423,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412969591","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9832049,0.00014334591,0.013549986,0.00006122079,0.000015347689,0.00017292582,0.000061599414,0.00004469272,0.0027458945],"genre_scores_gemma":[0.9954163,0.000029195739,0.0040591345,0.000022544758,0.000010326547,0.00016975202,0.00008431254,0.000013948466,0.00019453443],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9681037,0.019378236,0.0021109309,0.0036049872,0.0062206523,0.0005814502],"domain_scores_gemma":[0.8593172,0.103813864,0.011610255,0.011456145,0.012731849,0.0010705726],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04922393,0.0004996574,0.00051393517,0.002359252,0.00056656054,0.0020382972,0.0007004577,0.0007746584,0.00091930013],"category_scores_gemma":[0.12663516,0.0003418778,0.0009727496,0.001156365,0.0017064842,0.0018486036,0.003654448,0.0007623974,0.000271189],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000521718,0.00027153787,0.93834645,0.00008064509,0.0004402846,0.000057098907,0.004310286,0.0011426884,0.002505312,0.00054201996,0.00012869941,0.05165318],"study_design_scores_gemma":[0.0000492138,0.0015777091,0.97461843,0.00007102369,0.00027284856,0.0003153568,0.0021587485,0.014071969,0.0044251257,0.0016338214,0.0007606396,0.000045164954],"about_ca_topic_score_codex":0.0013086082,"about_ca_topic_score_gemma":0.0025168865,"teacher_disagreement_score":0.04922393,"about_ca_system_score_codex":0.00062868424,"about_ca_system_score_gemma":0.0006914506,"threshold_uncertainty_score":0.26032412},"labels":[],"label_agreement":null},{"id":"W4414209357","doi":"10.1177/02655322251348685","title":"Advancing language assessment for teaching and learning in the era of the artificial intelligence (AI) revolution: Promises and challenges","year":2025,"lang":"en","type":"article","venue":"Language Testing","topic":"Second Language Learning and Teaching","field":"Arts and Humanities","cited_by":10,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"Japan Society for the Promotion of Science; Waseda University","keywords":"Language assessment; Language proficiency; Language acquisition; Applications of artificial intelligence; Assessment for learning; Teaching method","score_opus":0.036526360261197204,"score_gpt":0.3059178929668351,"score_spread":0.26939153270563787,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414209357","genre_codex":"commentary","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034150075,0.05911489,0.10313363,0.7349624,0.004024211,0.00016850558,0.00032996194,0.0018832887,0.062233116],"genre_scores_gemma":[0.7340036,0.03911021,0.16406599,0.039334226,0.0066686557,0.00033281057,0.00045212402,0.0005935305,0.015438882],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9674701,0.019896409,0.0012984421,0.00124607,0.008777518,0.0013114583],"domain_scores_gemma":[0.7965822,0.1352032,0.0061294413,0.007563136,0.037814688,0.016707271],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0604332,0.0011005376,0.001660472,0.003185197,0.0025319373,0.015723525,0.004402491,0.0067512635,0.01120257],"category_scores_gemma":[0.14007701,0.00029895836,0.00047478208,0.001962687,0.01246689,0.030609287,0.011511995,0.010567287,0.0030343926],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025639584,0.0006985995,0.010416728,0.00083649653,0.00003727399,0.00012664245,0.0036536842,0.0012689616,0.0011096965,0.15699805,0.039322015,0.7852754],"study_design_scores_gemma":[0.00008860134,0.0006662869,0.007008088,0.0024171043,0.000029910525,0.00039903834,0.01094088,0.009736213,0.0031628094,0.77520335,0.19017716,0.0001705552],"about_ca_topic_score_codex":0.007479151,"about_ca_topic_score_gemma":0.0080650505,"teacher_disagreement_score":0.0604332,"about_ca_system_score_codex":0.006519475,"about_ca_system_score_gemma":0.021259658,"threshold_uncertainty_score":0.3196051},"labels":[],"label_agreement":null},{"id":"W7117366349","doi":"10.1177/02655322251406223","title":"Book review: Comprehensibility in language assessment: A broader perspective TavakoliP.CookeS.Comprehensibility in Language Assessment: A Broader Perspective. University of Toronto Press, 2024. 224 pp. ISBN 978 1 80050 434 9 (ePDF), 978 1 80050 433 2 (paperback), 978 1 80050 432 5 (hardback), £23.06 (Kindle), £24.27 (Paperback), £66.03 (Hardcover)","year":2025,"lang":"en","type":"article","venue":"Language Testing","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Perspective (graphical); Second language; On Language; Language assessment; Language proficiency","score_opus":0.025777876961311307,"score_gpt":0.36001450487924097,"score_spread":0.3342366279179297,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7117366349","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.000040716626,0.9873674,0.00014091466,0.006345189,0.004501574,0.000014649475,0.00007961717,0.000014062516,0.0014959759],"genre_scores_gemma":[0.0009211022,0.98072904,0.00030631604,0.0049440865,0.0056811925,0.00007178601,0.00019665045,0.000023547102,0.007126267],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9978606,0.0005623158,0.00030904834,0.00019771678,0.0009638557,0.000106451786],"domain_scores_gemma":[0.97912395,0.013283105,0.0015673302,0.00021537796,0.0052463454,0.00056388223],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028196736,0.0017686253,0.0041402173,0.00671943,0.0005306109,0.0036465675,0.002699778,0.0032892316,0.023432398],"category_scores_gemma":[0.017078552,0.0008271123,0.0012281745,0.008521168,0.0015145751,0.003989494,0.0014562975,0.004143673,0.01337608],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00005145434,0.00002301163,0.00012447065,0.011612424,0.00009020413,0.00007454197,0.00007229851,0.00009841464,0.00010878952,0.001090635,0.78471386,0.20193979],"study_design_scores_gemma":[0.00004891857,0.000090322865,0.0022003327,0.025783563,0.00021548262,0.0012917805,0.00015908979,0.000102550286,0.00012011373,0.002318604,0.96762496,0.00004432319],"about_ca_topic_score_codex":0.008398328,"about_ca_topic_score_gemma":0.016258681,"teacher_disagreement_score":0.023432398,"about_ca_system_score_codex":0.0024813511,"about_ca_system_score_gemma":0.0042071114,"threshold_uncertainty_score":0.07838923},"labels":[],"label_agreement":null}]}