{"meta":{"page":1,"per_page":50,"max_per_page":100,"total":57,"total_is_capped":false,"direct_labels_cover":0,"predictions_cover":57,"direct_label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline (scores rank; they never assert a category)","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12","author_layer_release":"2026-06-26"},"query_hash":"6e475cc8b99a","filters":{"venue":"Language Testing"}},"results":[{"id":"W2155682078","doi":"10.1191/0265532206lt337oa","title":"How assessing reading comprehension with multiple-choice questions shapes the construct: a cognitive processing perspective","year":2006,"lang":"en","type":"article","venue":"Language Testing","topic":"Reading and Literacy Development","field":"Psychology","cited_by":243,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Ottawa","funders":"","keywords":"Construct (python library); Reading comprehension; Psychology; Variety (cybernetics); Comprehension; Cognitive psychology; Cognition; Multiple choice; Perspective (graphical); Test (biology); Reading (process); Selection (genetic algorithm); Task (project management); Computer science; Linguistics; Artificial intelligence","authors":[{"name":"André Rupp","is_ca":false},{"name":"Tracy Ferne","is_ca":true},{"name":"Hyeran Choi","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02646085463337262,"gpt":0.3216621472035859,"spread":0.2952012925702133,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02734997,0.0006495494,0.0005401817,0.003295137,0.0005193373,0.007009767,0.001203516,0.001346379,0.001359612],"category_scores_gemma":[0.1107205,0.0005837109,0.0006254528,0.001870962,0.009179497,0.009850972,0.002226168,0.001989328,0.0002381781],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001215453,"about_ca_system_score_gemma":0.0007636339,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001248842,"about_ca_topic_score_gemma":0.001425639,"domain_scores_codex":[0.9756699,0.018743,0.0005953014,0.001773026,0.002835916,0.0003828058],"domain_scores_gemma":[0.8143643,0.168916,0.007079685,0.005206371,0.003885445,0.0005483833],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004850225,0.0008221492,0.1366886,0.00155044,0.00037915,0.0007410812,0.2894248,0.006169616,0.0366003,0.1701212,0.001154344,0.3558632],"study_design_scores_gemma":[0.0001798867,0.001616238,0.2452908,0.001030938,0.0003266637,0.002267285,0.07645,0.04680702,0.03385398,0.5738767,0.01776603,0.0005343136],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7053012,0.001653818,0.2672206,0.004521044,0.00006019053,0.0002155796,0.00009362078,0.0002420642,0.0206918],"genre_scores_gemma":[0.9506222,0.0005449234,0.04762226,0.0004274304,0.00004773659,0.0001621682,0.00004708039,0.00005740225,0.0004688047],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02734997,"threshold_uncertainty_score":0.1446422,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2134278381","doi":"10.1177/0265532207083743","title":"The key to success: English language testing in China","year":2008,"lang":"en","type":"article","venue":"Language Testing","topic":"Multilingual Education and Policy","field":"Social Sciences","cited_by":242,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"Ministry of Education, India; Ministry of Earth Sciences","keywords":"Language assessment; China; Context (archaeology); Psychology; English language; Test (biology); Linguistics; Test of English as a Foreign Language; Language proficiency; Chinese language; Mathematics education; History","authors":[{"name":"Liying Cheng","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0648421376261999,"gpt":0.4101903691881555,"spread":0.3453482315619556,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01096668,0.0002848776,0.0004112139,0.002144826,0.003949618,0.005014567,0.001094131,0.001288224,0.00429056],"category_scores_gemma":[0.02320735,0.0002176699,0.0001887109,0.004364367,0.008342939,0.004767886,0.003733133,0.001998505,0.0003395263],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.007463609,"about_ca_system_score_gemma":0.03821543,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.08723613,"about_ca_topic_score_gemma":0.06559477,"domain_scores_codex":[0.9929419,0.002335205,0.0004975031,0.0004724337,0.002289023,0.001464019],"domain_scores_gemma":[0.9778613,0.008931503,0.003178519,0.0007986427,0.003905594,0.005324371],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0001530115,0.0003269259,0.3462199,0.001287101,0.00004655093,0.003073614,0.04190045,0.001301723,0.001824679,0.1946201,0.03007205,0.3791739],"study_design_scores_gemma":[0.00006000548,0.0004483749,0.7214758,0.001610371,0.00007416737,0.0009092859,0.05818735,0.00280512,0.003204133,0.05148752,0.1595466,0.0001913222],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6194819,0.01485494,0.002672177,0.2125484,0.0006924016,0.0002303369,0.0002426285,0.0001161739,0.1491611],"genre_scores_gemma":[0.9883155,0.002913266,0.0006792504,0.003271846,0.0001216478,0.00005595314,0.00006687543,0.00001101544,0.004564708],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.08723613,"threshold_uncertainty_score":0.1734567,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2042225150","doi":"10.1191/0265532204lt273oa","title":"Evaluation of an in-depth vocabulary knowledge measure for assessing reading performance","year":2004,"lang":"en","type":"article","venue":"Language Testing","topic":"Second Language Acquisition and Learning","field":"Psychology","cited_by":237,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Test of English as a Foreign Language; Vocabulary; Reading comprehension; Reading (process); Context (archaeology); Measure (data warehouse); Test (biology); Psychology; Language proficiency; Sample (material); Vocabulary development; Mathematics education; Language assessment; Computer science; Linguistics; Teaching method","authors":[{"name":"David D. Qian","is_ca":false},{"name":"Mary Schedl","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.09731813955004215,"gpt":0.406059529447716,"spread":0.3087413898976738,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005204954,0.0004586938,0.0004321079,0.001784308,0.0003527467,0.0009910719,0.0008997726,0.0006165419,0.000809135],"category_scores_gemma":[0.0260272,0.0001744141,0.0004490964,0.0007387301,0.0003672121,0.001257441,0.0007220078,0.0006205713,0.0002609464],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001121133,"about_ca_system_score_gemma":0.001595734,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009828704,"about_ca_topic_score_gemma":0.03267641,"domain_scores_codex":[0.9951285,0.001304171,0.0005165302,0.0003042738,0.002522318,0.0002242592],"domain_scores_gemma":[0.9765419,0.01079436,0.004085685,0.001200044,0.006073779,0.001304202],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0007499933,0.001955929,0.8311982,0.0001957464,0.0001612497,0.000190213,0.001954512,0.001757206,0.01708511,0.0003511797,0.0005509937,0.1438496],"study_design_scores_gemma":[0.0000548773,0.003790393,0.9796784,0.00004666356,0.00006580561,0.0003343201,0.0009486981,0.005403274,0.008063748,0.0001833622,0.001395124,0.00003542263],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9945739,0.0001023362,0.002560387,0.00003932026,0.0000107754,0.0002456443,0.000177964,0.00002517858,0.002264597],"genre_scores_gemma":[0.9872411,0.0001366373,0.01083325,0.00003762513,0.00001269191,0.0003032234,0.0004538446,0.000007355715,0.000974141],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.009828704,"threshold_uncertainty_score":0.0275268,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2129988987","doi":"10.1177/0265532208092433","title":"Test review: College English Test (CET) in China","year":2008,"lang":"en","type":"article","venue":"Language Testing","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":192,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"","keywords":"Test (biology); College English; China; Psychology; Language assessment; Test of English as a Foreign Language; Mathematics education; Political science","authors":[{"name":"Ying Zheng","is_ca":true},{"name":"Liying Cheng","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01727451502830387,"gpt":0.2757766721564041,"spread":0.2585021571281003,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007446141,0.0007602825,0.00322565,0.004535535,0.0005979651,0.001192896,0.001594251,0.001191819,0.002900881],"category_scores_gemma":[0.03059378,0.0003188143,0.001048993,0.006943128,0.0008843123,0.0009492098,0.0007716336,0.0006229384,0.0004699427],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003094122,"about_ca_system_score_gemma":0.01092561,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.04074396,"about_ca_topic_score_gemma":0.0809317,"domain_scores_codex":[0.9954639,0.00139658,0.001563624,0.000360683,0.001082928,0.0001323273],"domain_scores_gemma":[0.9734994,0.01147348,0.003506493,0.0004757868,0.01022287,0.000822046],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001578689,0.0002392376,0.036948,0.1516278,0.002829411,0.0009439738,0.0004781037,0.0004283759,0.001136508,0.0007585466,0.1393269,0.6637045],"study_design_scores_gemma":[0.001853002,0.003364118,0.3141689,0.1312393,0.02105005,0.003093883,0.001565083,0.0008385925,0.002401056,0.0008900349,0.5193116,0.0002243224],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.01718771,0.9630475,0.0004798873,0.007553943,0.004514816,0.0004763713,0.002155648,0.00003385757,0.004550353],"genre_scores_gemma":[0.1609806,0.8112651,0.001503884,0.01353436,0.003208121,0.001000233,0.004036045,0.00004609154,0.004425582],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.04074396,"threshold_uncertainty_score":0.08101356,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2159609851","doi":"10.1191/0265532204lt288oa","title":"ESL/EFL instructors’ classroom assessment practices: purposes, methods, and procedures","year":2004,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":184,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Alberta; Queen's University","funders":"","keywords":"Psychology; English as a foreign language; Mathematics education; Tertiary level; Pedagogy; Second language; Language assessment; Linguistics","authors":[{"name":"Liying Cheng","is_ca":true},{"name":"Todd Rogers","is_ca":true},{"name":"Huiqin Hu","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06401151606627827,"gpt":0.4566933338742858,"spread":0.3926818178080075,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04419564,0.0005909574,0.0005557169,0.004196423,0.00248741,0.002311585,0.001412967,0.0005317276,0.002014607],"category_scores_gemma":[0.07329158,0.0004183087,0.0002716429,0.002983906,0.002312854,0.001303968,0.003125248,0.0007570963,0.001132535],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004144968,"about_ca_system_score_gemma":0.006230204,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01155767,"about_ca_topic_score_gemma":0.02301954,"domain_scores_codex":[0.9547526,0.02701685,0.005813431,0.002833513,0.007963849,0.001619732],"domain_scores_gemma":[0.9352966,0.02213053,0.006743305,0.006459848,0.02684863,0.00252109],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"qualitative","study_design_scores_codex":[0.000888226,0.002255439,0.2514513,0.0007885541,0.00002985677,0.0003699488,0.1596395,0.0004672911,0.0112482,0.001081218,0.00196139,0.5698192],"study_design_scores_gemma":[0.0003243159,0.002467649,0.7964455,0.001025194,0.00006302998,0.0005444374,0.1236496,0.00307432,0.03002802,0.001700874,0.04046831,0.0002087711],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9607155,0.0003306784,0.01445856,0.0004667196,0.00003704431,0.007025859,0.0004268307,0.0002383994,0.01630036],"genre_scores_gemma":[0.9315376,0.0003620019,0.052952,0.0002430971,0.00003468316,0.01053553,0.0002586859,0.00005263382,0.00402367],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04419564,"threshold_uncertainty_score":0.2337317,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2135663628","doi":"10.1177/0265532208097336","title":"Cognitive diagnostic assessment of L2 reading comprehension ability: Validity arguments for Fusion Model application to <i>LanguEdge</i> assessment","year":2008,"lang":"en","type":"article","venue":"Language Testing","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":175,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Reading comprehension; Psychology; Cognition; Test (biology); Profiling (computer programming); Dependability; Comprehension; Reading (process); Cognitive psychology; Computer science; Linguistics","authors":[{"name":"Eunice Eunhee Jang","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.4824119830634888,"gpt":0.5159648486704076,"spread":0.03355286560691884,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0486821,0.0007586219,0.001072731,0.004358161,0.0009653732,0.003803343,0.002019352,0.001754677,0.002137555],"category_scores_gemma":[0.2508255,0.0003459126,0.001366724,0.002192287,0.004630828,0.004936526,0.005208767,0.00225757,0.0003793943],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00236771,"about_ca_system_score_gemma":0.00219577,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003331222,"about_ca_topic_score_gemma":0.001666186,"domain_scores_codex":[0.9650055,0.02072003,0.001915497,0.002848978,0.008765366,0.0007446035],"domain_scores_gemma":[0.7637811,0.1912163,0.008795157,0.0173636,0.01699496,0.001848906],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.003420413,0.0008683964,0.4591054,0.0006854163,0.000779748,0.0006137151,0.01071717,0.02013513,0.007388768,0.07253753,0.003212869,0.4205355],"study_design_scores_gemma":[0.0003997541,0.002202476,0.1764929,0.0005914565,0.0005031759,0.001437463,0.005373914,0.6001076,0.01589247,0.1918741,0.004769902,0.0003548207],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7499437,0.0008948687,0.2258897,0.005164768,0.0001694286,0.0006953829,0.0004554502,0.000531794,0.01625488],"genre_scores_gemma":[0.9653413,0.00006738762,0.03382828,0.0002036996,0.00003523198,0.0002243116,0.00009728307,0.00001957974,0.0001829591],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0486821,"threshold_uncertainty_score":0.2574586,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2156770411","doi":"10.1177/0265532209104666","title":"Interacting in pairs in a test of oral proficiency: Co-constructing a better performance","year":2009,"lang":"en","type":"article","venue":"Language Testing","topic":"Multilingual Education and Policy","field":"Social Sciences","cited_by":163,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Toronto","funders":"","keywords":"Test (biology); Psychology; Meaning (existential); Context (archaeology); Negotiation; Social psychology; Mathematics education","authors":[{"name":"Lindsay Brooks","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06423787690980926,"gpt":0.4414124951436066,"spread":0.3771746182337974,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006801438,0.0005683058,0.0006489955,0.001443937,0.001426583,0.00430192,0.0007683398,0.0006234845,0.002579859],"category_scores_gemma":[0.03441317,0.0003851705,0.0004515966,0.0005162649,0.001963298,0.001413571,0.003850451,0.0009446254,0.0009158096],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007872855,"about_ca_system_score_gemma":0.0009887612,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002951451,"about_ca_topic_score_gemma":0.004255696,"domain_scores_codex":[0.990043,0.00591205,0.0005085882,0.0008335784,0.001994797,0.0007079415],"domain_scores_gemma":[0.9813172,0.007400581,0.004617434,0.001873934,0.002649,0.002141897],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0003848147,0.0009792591,0.7758493,0.0001003174,0.00009923329,0.001465103,0.1321388,0.0003792934,0.01267537,0.0004649543,0.0005889723,0.07487457],"study_design_scores_gemma":[0.0000282476,0.002172987,0.8458275,0.00005254036,0.00007729335,0.002120726,0.1358854,0.001743471,0.007694096,0.0009493894,0.003285721,0.0001625235],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9983104,0.00002696002,0.000435807,0.00005518669,0.000005901155,0.00001246824,0.0000126979,0.0000104587,0.001129929],"genre_scores_gemma":[0.9974848,0.0000316098,0.001389143,0.00002412083,0.000005125498,0.0000171096,0.00003400016,0.000009182167,0.001004926],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.006801438,"threshold_uncertainty_score":0.03596991,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2123012719","doi":"10.1177/026553220101800302","title":"Examining dialogue: another approach to content specification and to validating inferences drawn from test scores","year":2001,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":162,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Construct (python library); Psychology; Test (biology); Sociocultural evolution; Point (geometry); Cognition; Cognitive psychology; Content (measure theory); Mathematics education; Inference; Construct validity; Linguistics; Natural language processing; Social psychology; Computer science; Artificial intelligence; Psychometrics; Developmental psychology","authors":[],"retraction":null,"screen_n_in":null,"score":{"opus":0.2440808435611483,"gpt":0.2751833738551075,"spread":0.03110253029395921,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.161656,0.002729091,0.002763749,0.02730363,0.003611698,0.01562342,0.005862653,0.004393897,0.004174567],"category_scores_gemma":[0.4097496,0.000802374,0.002223392,0.01508754,0.009089683,0.01691436,0.009106664,0.005165908,0.001536712],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003753197,"about_ca_system_score_gemma":0.006489226,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00533489,"about_ca_topic_score_gemma":0.003941027,"domain_scores_codex":[0.7846398,0.1580065,0.01798476,0.0113404,0.02527799,0.002750428],"domain_scores_gemma":[0.3497685,0.4996819,0.03248017,0.05257063,0.06253923,0.002959656],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00109646,0.001500914,0.1126694,0.002429954,0.0007175228,0.001087532,0.109102,0.005130205,0.01836386,0.1303187,0.004889062,0.6126944],"study_design_scores_gemma":[0.0005741596,0.005203129,0.1279792,0.003305797,0.001004626,0.002155275,0.110943,0.09504624,0.0782926,0.459444,0.1147174,0.001334486],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1208789,0.0003642289,0.8447046,0.002643687,0.0002812423,0.00281024,0.001271952,0.001939338,0.02510589],"genre_scores_gemma":[0.3306119,0.0001701277,0.6593288,0.0008451003,0.0001841323,0.004014475,0.001191895,0.0004317374,0.003221876],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.161656,"threshold_uncertainty_score":0.8549289,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2105451781","doi":"10.1191/0265532204lt287oa","title":"Teacher formative assessment and talk in classroom contexts: assessment as discourse and assessment of discourse","year":2004,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":159,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Formative assessment; Psychology; Pedagogy; Discourse analysis; Applied linguistics; Mathematics education; Assessment for learning; Systemic functional linguistics; Linguistics","authors":[{"name":"Constant Leung","is_ca":false},{"name":"Bernard Mohan","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03032002706073061,"gpt":0.4286908790492009,"spread":0.3983708519884703,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03478556,0.000617564,0.0007696394,0.005578764,0.00147686,0.01191711,0.001677766,0.001498957,0.001622202],"category_scores_gemma":[0.1005457,0.0003586864,0.0003531172,0.003686693,0.01388191,0.009962572,0.005300997,0.002310413,0.0002805237],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00229644,"about_ca_system_score_gemma":0.003387991,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001555785,"about_ca_topic_score_gemma":0.002265697,"domain_scores_codex":[0.9522936,0.03846737,0.001698786,0.001208084,0.005936554,0.0003956448],"domain_scores_gemma":[0.9081358,0.07414794,0.007252437,0.00433499,0.004460025,0.001668874],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0001970836,0.0004703581,0.05181728,0.001429481,0.00009859348,0.000492739,0.4369698,0.00219146,0.006109028,0.1287884,0.001066504,0.3703692],"study_design_scores_gemma":[0.0001288726,0.001841601,0.1639504,0.003538027,0.0001551987,0.004283031,0.3164106,0.02057525,0.01895963,0.4026691,0.06704799,0.0004403103],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5576613,0.01008833,0.3531671,0.006669908,0.0003292862,0.0008942871,0.0001775124,0.000387094,0.07062522],"genre_scores_gemma":[0.9515393,0.001755937,0.04319308,0.0001705102,0.00008937758,0.0006560503,0.00004003309,0.00005014699,0.002505473],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03478556,"threshold_uncertainty_score":0.1839659,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2082767250","doi":"10.1177/0265532208101010","title":"An investigation into native and non-native teachers' judgments of oral English performance: A mixed methods approach","year":2009,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":137,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Toronto","funders":"","keywords":"Psychology; Pronunciation; Rasch model; Consistency (knowledge bases); Grammar; Mathematics education; First language; Internal consistency; Multimethodology; Linguistics; Psychometrics; Developmental psychology; Computer science; Artificial intelligence","authors":[{"name":"Younhee Kim","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04154220340015909,"gpt":0.3200407438186342,"spread":0.2784985404184751,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0417037,0.0006480066,0.0008543473,0.00311591,0.002603126,0.002956682,0.001306123,0.0006082811,0.001026302],"category_scores_gemma":[0.05004807,0.0006764524,0.0005815889,0.001619973,0.001699835,0.00136684,0.00185989,0.0005619381,0.0002681937],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001627646,"about_ca_system_score_gemma":0.002254832,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004566922,"about_ca_topic_score_gemma":0.01536712,"domain_scores_codex":[0.9665099,0.02428945,0.00224711,0.001928286,0.004367743,0.0006575187],"domain_scores_gemma":[0.9390922,0.04464059,0.004326257,0.003399726,0.007683879,0.0008573094],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.001390271,0.003530941,0.3214507,0.000972909,0.0003686852,0.0008487219,0.4738581,0.0004838374,0.01729441,0.001584493,0.0002813131,0.1779356],"study_design_scores_gemma":[0.0004271279,0.01309273,0.3884966,0.0005438776,0.0004753736,0.001454717,0.5517167,0.005656139,0.02621405,0.002588756,0.009041518,0.0002924264],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9879448,0.0001672156,0.009387945,0.00007053011,0.00001920398,0.0009306056,0.00007954667,0.00001685538,0.001383254],"genre_scores_gemma":[0.9609865,0.0002290241,0.0337419,0.0001724302,0.00002291856,0.003398923,0.0001208821,0.00002155667,0.001305829],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0417037,"threshold_uncertainty_score":0.2205529,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2116070328","doi":"10.1177/0265532210376379","title":"Think-aloud protocols in research on essay rating: An empirical study of their veridicality and reactivity","year":2010,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":121,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"York University","funders":"","keywords":"Think aloud protocol; Psychology; Protocol analysis; Perception; Empirical research; Sample (material); Qualitative research; Rating scale; Social psychology; Nomothetic and idiographic; Cognitive psychology; Applied psychology; Developmental psychology; Epistemology; Cognitive science","authors":[{"name":"Khaled Barkaoui","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2923764194054679,"gpt":0.5510859877479262,"spread":0.2587095683424582,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03743877,0.0007361121,0.0004927889,0.00120755,0.0009173541,0.001642874,0.001102275,0.0009466977,0.00116799],"category_scores_gemma":[0.2199901,0.000555056,0.000320532,0.001070857,0.001426186,0.001562504,0.001957292,0.001280929,0.0006609897],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004902453,"about_ca_system_score_gemma":0.0006471811,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001933144,"about_ca_topic_score_gemma":0.0002665671,"domain_scores_codex":[0.9395154,0.04742792,0.003363881,0.002915364,0.006367211,0.0004101973],"domain_scores_gemma":[0.6944738,0.2487216,0.0227628,0.01710609,0.01570473,0.001230987],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00291203,0.001885879,0.1145801,0.002211474,0.0003552911,0.0007919685,0.2230306,0.001853537,0.1190248,0.005478701,0.001786692,0.526089],"study_design_scores_gemma":[0.0007464606,0.02040762,0.5165713,0.002543318,0.0005755922,0.007781459,0.1235017,0.02727211,0.2092096,0.0289061,0.06154568,0.0009391778],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8939719,0.000418444,0.09813229,0.0001922338,0.0001273409,0.001318618,0.000135502,0.0002361061,0.005467661],"genre_scores_gemma":[0.913073,0.0005242794,0.07989421,0.0002689874,0.0001163691,0.003634382,0.0001949199,0.0001571192,0.002136816],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9625612,"threshold_uncertainty_score":0.1979975,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2136852860","doi":"10.1177/026553220101800303","title":"Native- and nonnative-speaking EFL teachers’ evaluation of Chinese students’ English writing","year":2001,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":117,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Psychology; Multivariate analysis of variance; First language; Foreign language; English as a foreign language; Point (geometry); Language assessment; Language proficiency; Linguistics; Mathematics education","authors":[{"name":"Ling Shi","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06659189121337757,"gpt":0.3396810487605434,"spread":0.2730891575471659,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001973909,0.0002927629,0.0002880072,0.0008390917,0.0005597821,0.0006305817,0.0001919723,0.0002221942,0.001587743],"category_scores_gemma":[0.009412385,0.0001383219,0.0002044293,0.0003172577,0.0005758541,0.0003604811,0.0006207736,0.0002465589,0.0003641356],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004430518,"about_ca_system_score_gemma":0.0004686286,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003380874,"about_ca_topic_score_gemma":0.007788473,"domain_scores_codex":[0.9990682,0.0002329998,0.0001481355,0.0001041918,0.0003455208,0.0001009122],"domain_scores_gemma":[0.9929945,0.002127946,0.001564083,0.0003610742,0.001807736,0.001144737],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0009907834,0.0005527464,0.8243179,0.0002250489,0.0001022376,0.0007832336,0.07450287,0.0002794949,0.04859459,0.0000966666,0.0003880839,0.0491664],"study_design_scores_gemma":[0.0000319511,0.0006257896,0.974772,0.00001409444,0.00001449679,0.0003294568,0.01972574,0.000351763,0.003464028,0.0000422883,0.0006046671,0.00002365589],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9995974,0.00001371893,0.0000257227,0.000004236887,0.000001238388,0.000003916825,0.00000559113,0.000001240382,0.0003470033],"genre_scores_gemma":[0.9992281,0.0000300655,0.0000815543,0.000007730022,0.000002611948,0.00001089476,0.00002837452,0.000001664512,0.0006089127],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.003380874,"threshold_uncertainty_score":0.01043916,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2122362059","doi":"10.1191/0265532206lt322oa","title":"Aiming for positive washback: a case study of international teaching assistants","year":2005,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":102,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université Laval","funders":"","keywords":"Test (biology); Psychology; Language proficiency; Process (computing); Mathematics education; Empirical research; Computer science","authors":[{"name":"Shahrzad Saif","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05418180584152639,"gpt":0.417552964658704,"spread":0.3633711588171776,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007090907,0.0009132704,0.0006940333,0.001177574,0.007476408,0.002904533,0.002595777,0.003783207,0.00242489],"category_scores_gemma":[0.02606587,0.0007124443,0.0006381013,0.0009941871,0.002688632,0.00161748,0.003125132,0.004107584,0.0005478734],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002406489,"about_ca_system_score_gemma":0.002727034,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004929525,"about_ca_topic_score_gemma":0.0130035,"domain_scores_codex":[0.9923446,0.004435326,0.0003190832,0.0004494324,0.000847509,0.001604164],"domain_scores_gemma":[0.9843876,0.008667849,0.001786283,0.0007516964,0.001098053,0.003308472],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.0006586729,0.01561258,0.09269043,0.0005907764,0.00008675403,0.09080338,0.6864872,0.001088879,0.006645918,0.002685027,0.001792372,0.1008581],"study_design_scores_gemma":[0.000175311,0.006551826,0.04126441,0.000324124,0.00007951962,0.03349762,0.8831282,0.002379781,0.008789723,0.00160231,0.02205646,0.0001506972],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9960716,0.00008090044,0.001216433,0.0005325318,0.00001895767,0.000138884,0.00001137015,0.00001623256,0.001913105],"genre_scores_gemma":[0.992968,0.0002593973,0.003330395,0.0003558495,0.00003388179,0.0001489004,0.00001667783,0.00001965531,0.002867276],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007476408,"threshold_uncertainty_score":0.0375008,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2109863812","doi":"10.1191/0265532204lt278oa","title":"A teacher-verification study of speaking and writing prototype tasks for a new TOEFL","year":2004,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":93,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Test of English as a Foreign Language; Active listening; Psychology; Formative assessment; Mathematics education; Test (biology); CLARITY; Reading (process); Second language writing; Language proficiency; Presentation (obstetrics); Language assessment; Pedagogy; Computer science; Second language; Linguistics; Communication","authors":[{"name":"Alister Cumming","is_ca":true},{"name":"Leslie Grant","is_ca":false},{"name":"Patricia Mulcahy-Ernt","is_ca":false},{"name":"Donald E. Powers","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08046825576428888,"gpt":0.3114058708120012,"spread":0.2309376150477123,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01861471,0.0006208867,0.0009603996,0.001292594,0.001853008,0.001735428,0.001565011,0.0009866271,0.001701944],"category_scores_gemma":[0.08789999,0.0007758045,0.0004841319,0.0005552672,0.001156927,0.001861287,0.00123315,0.001889018,0.0007459284],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001631151,"about_ca_system_score_gemma":0.001822153,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003271908,"about_ca_topic_score_gemma":0.009383092,"domain_scores_codex":[0.992465,0.004347793,0.0007357041,0.0007294015,0.001326899,0.0003952288],"domain_scores_gemma":[0.8910272,0.07368827,0.006842709,0.009079237,0.01620084,0.003161734],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"observational","study_design_scores_codex":[0.002855938,0.03294211,0.2388501,0.0008384478,0.00007237328,0.002615521,0.4407515,0.001073167,0.04442396,0.0006428777,0.002023237,0.2329106],"study_design_scores_gemma":[0.0018267,0.05766236,0.614072,0.0003974746,0.0001408935,0.005233862,0.2254458,0.01058414,0.05340882,0.001231481,0.02955977,0.0004366987],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.998639,0.00001713757,0.0007689169,0.00002716865,0.000008696666,0.0001215204,0.0000150893,0.00001553825,0.0003868745],"genre_scores_gemma":[0.9916943,0.00006048545,0.0055071,0.0001148041,0.00001614463,0.0003316253,0.0001041072,0.00002439609,0.002147121],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01861471,"threshold_uncertainty_score":0.09844518,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2078562183","doi":"10.1177/0265532210384253","title":"Impact and consequences of school-based assessment (SBA): Students’ and parents’ views of SBA in Hong Kong","year":2011,"lang":"en","type":"article","venue":"Language Testing","topic":"Parental Involvement in Education","field":"Social Sciences","cited_by":92,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"","keywords":"Perception; Psychology; Competence (human resources); Certificate; Context (archaeology); Medical education; Developmental psychology; Mathematics education; Pedagogy; Social psychology; Medicine","authors":[{"name":"Liying Cheng","is_ca":true},{"name":"Stephen Andrews","is_ca":false},{"name":"Ying Yu","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2119144028794248,"gpt":0.4574731512778527,"spread":0.2455587483984279,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002833808,0.0002204144,0.0002690238,0.000586298,0.0009681106,0.001600656,0.000314239,0.0002910621,0.001092497],"category_scores_gemma":[0.006363724,0.0002712407,0.0004349492,0.0005089076,0.0009212744,0.0005836313,0.001182645,0.0007053263,0.0001459012],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001486777,"about_ca_system_score_gemma":0.001363745,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.04828833,"about_ca_topic_score_gemma":0.06129435,"domain_scores_codex":[0.9978487,0.001014085,0.0002646855,0.0001213088,0.0004361683,0.0003150156],"domain_scores_gemma":[0.9920871,0.001995311,0.002901582,0.0002333637,0.001143552,0.001639245],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"qualitative","study_design_scores_codex":[0.00007075116,0.0001370455,0.9627547,0.00003491764,0.00003913721,0.0005007365,0.02943473,0.0000865023,0.0004515166,0.00006465663,0.0001492746,0.006275937],"study_design_scores_gemma":[0.000003343356,0.0001634606,0.9519075,0.00002465955,0.00001782094,0.0001287974,0.04681305,0.0001411607,0.0002283969,0.00001930644,0.00053797,0.0000144894],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9995866,0.00003414356,0.00001261457,0.00004259589,0.000001854862,0.000002450719,0.00001043425,6.366246e-7,0.0003085792],"genre_scores_gemma":[0.9997705,0.00004332539,0.00002023794,0.00001188314,8.951129e-7,0.000002909451,0.00001076562,4.327843e-7,0.0001391036],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04828833,"threshold_uncertainty_score":0.0960145,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2096343999","doi":"10.1191/0265532203lt248oa","title":"Does item-level DIF manifest itself in scale-level analyses? Implications for translating language tests","year":2003,"lang":"en","type":"article","venue":"Language Testing","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":92,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of British Columbia","funders":"","keywords":"Differential item functioning; Equivalence (formal languages); Psychology; Scale (ratio); Measurement invariance; Item response theory; Item analysis; Statistics; Test (biology); Psychometrics; Developmental psychology; Linguistics; Structural equation modeling; Mathematics; Confirmatory factor analysis","authors":[{"name":"Bruno D. Zumbo","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.6773012489253456,"gpt":0.5390986227521983,"spread":0.1382026261731473,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1741679,0.001668009,0.00250879,0.003758,0.002100031,0.006615139,0.003071767,0.002287974,0.003827184],"category_scores_gemma":[0.665731,0.0009926538,0.002300151,0.004827591,0.01132217,0.01101788,0.003984601,0.004101844,0.0009048782],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003611376,"about_ca_system_score_gemma":0.003728779,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002854875,"about_ca_topic_score_gemma":0.002577238,"domain_scores_codex":[0.7531396,0.2154706,0.009110659,0.005530565,0.01497175,0.001776866],"domain_scores_gemma":[0.389396,0.5342252,0.01715731,0.03871311,0.01906081,0.001447509],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002117026,0.0009511592,0.2104652,0.002371002,0.0007993317,0.002997644,0.04360043,0.01866808,0.005589718,0.1542625,0.005460616,0.5527174],"study_design_scores_gemma":[0.0006893126,0.002445092,0.1322249,0.002315832,0.0004634399,0.002301843,0.04180646,0.09224727,0.01549298,0.6971639,0.01245412,0.0003947009],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4390763,0.003787065,0.4727606,0.04856186,0.001222171,0.001585674,0.0006147871,0.001363012,0.03102848],"genre_scores_gemma":[0.8519405,0.0006638839,0.1421973,0.002825819,0.0002853731,0.0008046106,0.0001775433,0.0002399375,0.0008651118],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1741679,"threshold_uncertainty_score":0.9210988,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2776964473","doi":"10.1177/0265532217716732","title":"The development of EFL examinations in Haiti: Collaboration and language assessment literacy development","year":2017,"lang":"en","type":"article","venue":"Language Testing","topic":"Multilingual Education and Policy","field":"Social Sciences","cited_by":87,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"McGill University; University of Ottawa","funders":"","keywords":"Christian ministry; Psychology; Literacy; Medical education; Professional development; English language; Pedagogy; Faculty development; Mathematics education; Language assessment; Language development; Political science; Medicine","authors":[{"name":"Beverly Baker","is_ca":true},{"name":"Caroline Riches","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.07883816509620944,"gpt":0.5035994185271453,"spread":0.4247612534309358,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04317266,0.0002214107,0.0003597841,0.001689405,0.01084324,0.004123246,0.001911733,0.001404432,0.001780484],"category_scores_gemma":[0.04967014,0.0007273566,0.0001777923,0.001272293,0.003539521,0.003543017,0.01041762,0.001934011,0.0004172119],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.009886752,"about_ca_system_score_gemma":0.04395098,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03409944,"about_ca_topic_score_gemma":0.06795359,"domain_scores_codex":[0.9661732,0.02754051,0.0009018516,0.001046697,0.001380276,0.002957417],"domain_scores_gemma":[0.9612841,0.0194648,0.003678821,0.001671457,0.006207282,0.007693551],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"observational","study_design_scores_codex":[0.0001823145,0.00146678,0.101526,0.0002361681,0.00001742455,0.00242675,0.7636209,0.0003720589,0.001965433,0.002737127,0.001373695,0.1240754],"study_design_scores_gemma":[0.00007261232,0.0007967119,0.08785702,0.0004099426,0.00001982711,0.0006328819,0.8857672,0.001188861,0.002187156,0.00171882,0.01928093,0.00006793575],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9862373,0.00016605,0.001905393,0.003014907,0.00002089887,0.0006132697,0.0000338934,0.00002656346,0.007981763],"genre_scores_gemma":[0.9932362,0.0001463185,0.004686413,0.0002985017,0.000004999048,0.0003460521,0.00002697889,0.000007432537,0.001247178],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04317266,"threshold_uncertainty_score":0.2283216,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2104209819","doi":"10.1177/0265532212436659","title":"Topical knowledge and ESL writing","year":2012,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":77,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of British Columbia","funders":"","keywords":"Impromptu; Cohesion (chemistry); Language proficiency; Psychology; Test (biology); Mathematics education; Language assessment; Second language writing; Test of English as a Foreign Language; English for academic purposes; Task (project management); Pedagogy; Second language; Linguistics; Computer science","authors":[{"name":"Ling He","is_ca":true},{"name":"Ling Shi","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06154722270013897,"gpt":0.2921705855996226,"spread":0.2306233628994836,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001752398,0.0003034307,0.0003546436,0.001174697,0.0005090172,0.001928227,0.0003944389,0.0003298881,0.00416008],"category_scores_gemma":[0.04242016,0.0001303335,0.0001709647,0.0006123751,0.00110405,0.001070592,0.001222518,0.0006481987,0.000538443],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005228678,"about_ca_system_score_gemma":0.0005677603,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001437442,"about_ca_topic_score_gemma":0.00199298,"domain_scores_codex":[0.9973273,0.0007975129,0.0002076453,0.0003107834,0.001169985,0.0001867091],"domain_scores_gemma":[0.9495186,0.03075438,0.01112779,0.001907143,0.003928151,0.002764012],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0008432537,0.002231296,0.7010803,0.0002693788,0.0001237585,0.001686136,0.01276344,0.0007439542,0.02071556,0.0008683637,0.0009654376,0.2577091],"study_design_scores_gemma":[0.00002660655,0.0009517952,0.983187,0.0000791755,0.00005342859,0.0009848257,0.00551118,0.0008244693,0.004980819,0.001491023,0.001876488,0.00003329484],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9930372,0.0002000144,0.0004136902,0.00007272478,0.000006227482,0.00001000934,0.00002653801,0.00001703191,0.006216513],"genre_scores_gemma":[0.9978923,0.0001324444,0.0003998489,0.00001951315,0.00000897995,0.000009183703,0.00004872256,0.000008146099,0.001480901],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.00416008,"threshold_uncertainty_score":0.01391685,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2005408543","doi":"10.1191/0265532204lt292oa","title":"Test decisions over time: tracking validity","year":2004,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":69,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Carleton University","funders":"","keywords":"Test (biology); Psychology; Sample (material); Active listening; Test validity; Applied psychology; Social psychology; Mathematics education; Psychometrics; Developmental psychology","authors":[{"name":"Janna Fox","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1004204092065655,"gpt":0.2939022965666889,"spread":0.1934818873601234,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1224852,0.0005251965,0.0007695573,0.005914486,0.002040442,0.004325276,0.002666495,0.001454551,0.001409485],"category_scores_gemma":[0.4049832,0.0006602926,0.00139995,0.005900171,0.003730278,0.005367981,0.006512977,0.002184071,0.0005205143],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003582884,"about_ca_system_score_gemma":0.004035691,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01571751,"about_ca_topic_score_gemma":0.01146669,"domain_scores_codex":[0.9097165,0.04730969,0.008921941,0.009625874,0.02177192,0.002654027],"domain_scores_gemma":[0.4533105,0.3706812,0.08062076,0.05320383,0.03950938,0.002674302],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0003391801,0.0002117839,0.921078,0.0001258579,0.0002026776,0.00007472829,0.01153623,0.001216883,0.0005204959,0.003101398,0.0004475926,0.06114526],"study_design_scores_gemma":[0.00008925429,0.0009019149,0.9472898,0.0002719667,0.0001649461,0.0002626238,0.008427924,0.01603126,0.00411244,0.01288039,0.00943786,0.0001296748],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9551107,0.0004793142,0.02808628,0.0007952903,0.0001213139,0.001084961,0.0008830409,0.0001678499,0.01327132],"genre_scores_gemma":[0.9877666,0.000112852,0.008922825,0.0001366039,0.00003814238,0.001001881,0.0006341611,0.00005267294,0.001334183],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1224852,"threshold_uncertainty_score":0.6477714,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2145055698","doi":"10.1177/0265532210368717","title":"Explaining ESL essay holistic scores: A multilevel modeling approach","year":2010,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":64,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"York University","funders":"Educational Testing Service","keywords":"Psychology; Argumentation theory; Context (archaeology); Multilevel model; Set (abstract data type); Multilevel modelling; Social psychology; Epistemology; Statistics; Computer science","authors":[{"name":"Khaled Barkaoui","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1195594706556013,"gpt":0.2960164806540859,"spread":0.1764570099984846,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02143838,0.001149958,0.001252039,0.003385942,0.00127685,0.002500051,0.001755244,0.0008932617,0.003217856],"category_scores_gemma":[0.05999744,0.0005565675,0.00319727,0.002971543,0.0007871265,0.001354862,0.002714962,0.002110571,0.0005126999],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00121941,"about_ca_system_score_gemma":0.001244723,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01274714,"about_ca_topic_score_gemma":0.009672363,"domain_scores_codex":[0.9846848,0.01102033,0.0006814038,0.001720024,0.001386587,0.0005068664],"domain_scores_gemma":[0.9512467,0.03797689,0.003823817,0.003778102,0.002673199,0.0005012836],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000445789,0.0004304921,0.8650311,0.0002404689,0.002834687,0.0003303559,0.007359024,0.01943192,0.001955191,0.01132918,0.001978673,0.08863313],"study_design_scores_gemma":[0.000104547,0.001206616,0.4178872,0.0002116721,0.001107979,0.0003594108,0.003312034,0.5430786,0.002164633,0.02590172,0.004474832,0.0001907043],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7655522,0.0002507368,0.2289283,0.0006523838,0.00006473838,0.000614025,0.001429692,0.0003741534,0.002133884],"genre_scores_gemma":[0.9431535,0.00006960533,0.05454081,0.00004303349,0.00002640606,0.0007340079,0.0006876765,0.00005312483,0.000691813],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02143838,"threshold_uncertainty_score":0.1133783,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2100317885","doi":"10.1177/0265532215570924","title":"How do young students with different profiles of reading skill mastery, perceived ability, and goal orientation respond to holistic diagnostic feedback?","year":2015,"lang":"en","type":"article","venue":"Language Testing","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":61,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Psychology; Reading (process); Goal orientation; Cognition; Developmental psychology; Orientation (vector space); Mathematics education; Cognitive psychology; Social psychology","authors":[{"name":"Eunice Eunhee Jang","is_ca":true},{"name":"Maggie Dunlop","is_ca":true},{"name":"Gina Park","is_ca":true},{"name":"Edith H. van der Boom","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06663303109817183,"gpt":0.3901718268217531,"spread":0.3235387957235813,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009951954,0.0001684515,0.0003243017,0.0007574128,0.0001962042,0.001320397,0.0002192806,0.0005224666,0.0008466254],"category_scores_gemma":[0.005394971,0.0002146148,0.0003288368,0.0003602291,0.0003536002,0.0009657927,0.0004940955,0.0004194899,0.0004206041],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000286605,"about_ca_system_score_gemma":0.0003040011,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002819832,"about_ca_topic_score_gemma":0.004370326,"domain_scores_codex":[0.9995756,0.00007421904,0.00004452172,0.00009334303,0.000124324,0.00008803581],"domain_scores_gemma":[0.9974926,0.000545133,0.001184924,0.0001125713,0.0002843869,0.0003803272],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00008681572,0.0001199401,0.9770969,0.00002816981,0.00003437372,0.0001138761,0.004708385,0.00005817071,0.002588935,0.00003944948,0.00007450683,0.01505051],"study_design_scores_gemma":[0.000003126467,0.0001372215,0.9938287,0.00001048081,0.00001113352,0.0001515604,0.004872378,0.0001491179,0.000601974,0.00006990036,0.0001554969,0.000008960192],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9996716,0.00004553141,0.00006525503,0.00001999343,0.000001689593,0.000003214441,0.00001528497,0.000001893202,0.0001754968],"genre_scores_gemma":[0.9995452,0.00006948855,0.00009472314,0.00002962061,0.000001610482,0.000005583947,0.00004685948,0.000001658761,0.0002051712],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.002819832,"threshold_uncertainty_score":0.00560683,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2611914815","doi":"10.1177/0265532217703433","title":"Developing a user-oriented second language comprehensibility scale for English-medium universities","year":2017,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":59,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Alberta; Concordia University","funders":"","keywords":"Operationalization; Formative assessment; Psychology; Construct (python library); English for academic purposes; Scale (ratio); Language proficiency; Point (geometry); Task (project management); Focus (optics); Rating scale; Mathematics education; Linguistics; Computer science","authors":[{"name":"Talia Isaacs","is_ca":false},{"name":"Pavel Trofimovich","is_ca":true},{"name":"Jennifer A. Foote","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04907756014366304,"gpt":0.2904557913624791,"spread":0.241378231218816,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01177776,0.0005527568,0.0005555551,0.002307403,0.0006274502,0.002027158,0.001117223,0.0008093312,0.00221977],"category_scores_gemma":[0.03207162,0.0004758062,0.001373918,0.0007732368,0.0005265892,0.002107647,0.002295832,0.001628323,0.0009028001],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008873254,"about_ca_system_score_gemma":0.001726695,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009130664,"about_ca_topic_score_gemma":0.0015929,"domain_scores_codex":[0.9965264,0.001046375,0.0006784603,0.0002134978,0.001347456,0.0001877615],"domain_scores_gemma":[0.980248,0.009440136,0.001852097,0.0009450946,0.006638827,0.0008756489],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0006056569,0.002179472,0.5337923,0.0009707609,0.0002595792,0.0005771188,0.05831706,0.002789708,0.02914811,0.004516875,0.00966571,0.3571776],"study_design_scores_gemma":[0.0001501068,0.0031439,0.9150583,0.0005246436,0.00009141779,0.0006516168,0.02177642,0.01435891,0.01218583,0.004114476,0.02767676,0.000267712],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.9569821,0.000103177,0.02978731,0.0003227215,0.00009236213,0.004272165,0.0007172226,0.0003495421,0.007373238],"genre_scores_gemma":[0.8324381,0.0002245594,0.1517501,0.0001585163,0.00003049364,0.01059772,0.00168246,0.00008426433,0.003033924],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.01177776,"threshold_uncertainty_score":0.06228751,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4251941390","doi":"10.1177/0265532220929918","title":"Automated scoring of junior and senior high essays using Coh-Metrix features: Implications for large-scale language testing","year":2020,"lang":"en","type":"article","venue":"Language Testing","topic":"Software Engineering Research","field":"Computer Science","cited_by":58,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Centre for Advancing Health Outcomes; University of Alberta","funders":"","keywords":"Natural language processing; Artificial intelligence; Rating scale; Computer science; Scale (ratio); Quality (philosophy); Construct (python library); Psychology; Computational linguistics; Disadvantaged; Machine learning; Developmental psychology","authors":[{"name":"Syed Latifi","is_ca":true},{"name":"Mark J. Gierl","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0384669575825602,"gpt":0.314934657209404,"spread":0.2764676996268438,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02200578,0.000703687,0.0005234812,0.001890332,0.0006181063,0.001886702,0.001137919,0.0005495879,0.001673234],"category_scores_gemma":[0.1303999,0.0002520036,0.0004401214,0.001925125,0.0007089208,0.002083595,0.001714188,0.001215067,0.0006363264],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001110748,"about_ca_system_score_gemma":0.001674405,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00415322,"about_ca_topic_score_gemma":0.009832878,"domain_scores_codex":[0.9826245,0.01114907,0.001145939,0.001377271,0.003304211,0.0003989929],"domain_scores_gemma":[0.8740447,0.0725709,0.01356838,0.01436639,0.02334865,0.002100986],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000787052,0.0008156315,0.5155722,0.0002352969,0.0001439789,0.0001213795,0.003062134,0.01071243,0.00787936,0.002254873,0.006686268,0.4517294],"study_design_scores_gemma":[0.00013235,0.001726347,0.665396,0.0001627556,0.00006515012,0.000316762,0.003222946,0.2953317,0.01865038,0.006333066,0.008492175,0.0001705095],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9057022,0.0001547389,0.08502079,0.000737787,0.0000782409,0.0006546102,0.0009435304,0.00160238,0.005105739],"genre_scores_gemma":[0.9451553,0.00002860357,0.05261295,0.00005831062,0.00002943647,0.0003828008,0.0007025354,0.00008279576,0.000947226],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02200578,"threshold_uncertainty_score":0.1163791,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2145523583","doi":"10.1177/026553220101800206","title":"ESL/EFL instructors' practices for writing assessment: specific purposes or general purposes?","year":2001,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":57,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Toronto","funders":"","keywords":"Interview; Psychology; Mathematics education; English for academic purposes; Pedagogy; Writing assessment; Language assessment; Process (computing); Sociology; Computer science","authors":[{"name":"Alister Cumming","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1428657225810786,"gpt":0.3633483596418732,"spread":0.2204826370607945,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01512459,0.000282616,0.0003907528,0.001188716,0.001787527,0.003257473,0.001028825,0.0008897464,0.0009664996],"category_scores_gemma":[0.05630838,0.000318939,0.0002231019,0.001010821,0.002941526,0.003274792,0.002297257,0.00161694,0.0003583101],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0013912,"about_ca_system_score_gemma":0.002267867,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001578454,"about_ca_topic_score_gemma":0.00378765,"domain_scores_codex":[0.9851059,0.009808451,0.001139318,0.0009636215,0.002074424,0.0009081899],"domain_scores_gemma":[0.9687103,0.01576947,0.004708163,0.00242899,0.005970492,0.002412532],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"qualitative","study_design_scores_codex":[0.00005460405,0.0001956144,0.1068,0.0002751894,0.0000147884,0.0005099837,0.7756333,0.0001571036,0.004662677,0.004665891,0.001443817,0.1055869],"study_design_scores_gemma":[0.00003726019,0.0003762601,0.1530941,0.0007481105,0.00002624354,0.001507744,0.7893734,0.001909571,0.003915997,0.005431462,0.04346956,0.0001103441],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9754786,0.0003975091,0.01141316,0.002327866,0.00005320139,0.0001512855,0.00002671045,0.00006892504,0.01008273],"genre_scores_gemma":[0.99277,0.0001801102,0.004997498,0.000271631,0.00001682254,0.0001413656,0.00001844971,0.00002392397,0.00158016],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01512459,"threshold_uncertainty_score":0.07998741,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2137984677","doi":"10.1177/0265532207071510","title":"A confirmatory approach to differential item functioning on an ESL reading assessment","year":2006,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":57,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Alberta Advanced Education","funders":"","keywords":"Differential item functioning; Psychology; Reading (process); Test (biology); Language proficiency; Item response theory; Flagging; Cognitive psychology; Psychometrics; Developmental psychology; Linguistics; Mathematics education","authors":[{"name":"Marilyn L. Abbott","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04018946173988295,"gpt":0.2719887309013568,"spread":0.2317992691614739,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.09328806,0.002526378,0.001445898,0.01143738,0.00257325,0.002925877,0.001870863,0.001022589,0.002979851],"category_scores_gemma":[0.2107984,0.0006834014,0.001965057,0.004800066,0.002223623,0.002750825,0.003253985,0.002722831,0.001002356],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001960218,"about_ca_system_score_gemma":0.005023564,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006755117,"about_ca_topic_score_gemma":0.0094355,"domain_scores_codex":[0.9245896,0.0522566,0.005226041,0.005923046,0.01100552,0.0009991938],"domain_scores_gemma":[0.7379405,0.1397043,0.0157613,0.03585742,0.06920537,0.001531087],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0009849996,0.001035889,0.3387326,0.001720917,0.0009467673,0.001017255,0.03804481,0.003044474,0.01680476,0.07664543,0.006040409,0.5149817],"study_design_scores_gemma":[0.0008131514,0.006222638,0.4699576,0.00280176,0.002017147,0.004852135,0.04737697,0.1150345,0.04214418,0.2624544,0.04547226,0.0008532064],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1889524,0.0005467553,0.7808871,0.001841866,0.0002917139,0.005390795,0.0009454471,0.0009506812,0.02019334],"genre_scores_gemma":[0.5546777,0.0002023282,0.4369658,0.0005771914,0.00009771049,0.005045694,0.0008496309,0.0001162743,0.001467605],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.09328806,"threshold_uncertainty_score":0.4933603,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2027399021","doi":"10.1177/0265532207076363","title":"The challenges of the Ontario Secondary School Literacy Test for second language students","year":2007,"lang":"en","type":"article","venue":"Language Testing","topic":"Second Language Acquisition and Learning","field":"Psychology","cited_by":57,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"Queen's University","funders":"","keywords":"Literacy; Test (biology); Psychology; Mathematics education; Vocabulary; Context (archaeology); Reading (process); English as a second language; Language proficiency; Language assessment; Pedagogy; Linguistics","authors":[{"name":"Liying Cheng","is_ca":true},{"name":"Don A. Klinger","is_ca":true},{"name":"Ying Zheng","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02266960080524535,"gpt":0.3419876094318603,"spread":0.319318008626615,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001796755,0.0003662097,0.0004718106,0.001136213,0.001441787,0.0015697,0.0007051223,0.0004226475,0.002210685],"category_scores_gemma":[0.01481812,0.0001956697,0.0003705626,0.001220336,0.0006230456,0.000473904,0.001089984,0.0004886069,0.0007555356],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004974371,"about_ca_system_score_gemma":0.01111162,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.5516672,"about_ca_topic_score_gemma":0.7850741,"domain_scores_codex":[0.9976742,0.0002506638,0.0001910837,0.0001413319,0.001440644,0.0003020704],"domain_scores_gemma":[0.9934663,0.001206819,0.0008911825,0.000257354,0.002795786,0.001382542],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0002638228,0.0003008131,0.9281842,0.00005953989,0.00002687856,0.0003953329,0.003645854,0.0001857842,0.002151959,0.0004579008,0.006141947,0.05818605],"study_design_scores_gemma":[0.00001634382,0.0001283059,0.9943805,0.00001775519,0.000007187971,0.00009296866,0.001107603,0.0003258138,0.0004552353,0.00009974081,0.003358815,0.000009631814],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9835206,0.0002173342,0.000281047,0.0005691026,0.00003835203,0.0001084047,0.0005088436,0.00003546198,0.01472085],"genre_scores_gemma":[0.9949885,0.0001166404,0.0006515359,0.00008861466,0.000009011625,0.00007516584,0.000652091,0.00001155566,0.003406747],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9950256,"threshold_uncertainty_score":0.9019463,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3040446934","doi":"10.1177/0265532220937830","title":"More efficient processes for creating automated essay scoring frameworks: A demonstration of two algorithms","year":2020,"lang":"en","type":"article","venue":"Language Testing","topic":"Topic Modeling","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Rubric; Artificial intelligence; Machine learning; Computer science; Support vector machine; Convolutional neural network; Feature engineering; Deep learning; Artificial neural network; Natural language processing; Meaning (existential); Feature (linguistics); Strengths and weaknesses; F1 score; Algorithm; Mathematics; Mathematics education; Linguistics; Psychology","authors":[{"name":"Jinnie Shin","is_ca":true},{"name":"Mark J. Gierl","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0348837319221498,"gpt":0.309711483163347,"spread":0.2748277512411972,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008945438,0.001384207,0.00104886,0.002867866,0.0006793856,0.003223549,0.002265602,0.001473953,0.005215978],"category_scores_gemma":[0.03084167,0.0006046362,0.0009189309,0.00137674,0.0008801701,0.004060037,0.003932704,0.002276634,0.003369427],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001126093,"about_ca_system_score_gemma":0.001913853,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0040064,"about_ca_topic_score_gemma":0.003722943,"domain_scores_codex":[0.9912093,0.003265555,0.0008352937,0.001474984,0.002880558,0.0003344095],"domain_scores_gemma":[0.9843061,0.005490905,0.0009894988,0.003672051,0.004896226,0.0006451946],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003645668,0.0004658359,0.005595087,0.0001998245,0.00009232007,0.000173118,0.0008095588,0.02525309,0.01872413,0.02787673,0.00933248,0.9111131],"study_design_scores_gemma":[0.00009948329,0.0002208056,0.004040509,0.00006183771,0.00003269862,0.0003489883,0.0003075796,0.9240695,0.0313393,0.02094692,0.01842266,0.0001096736],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01761846,0.0001480532,0.9709029,0.0003394759,0.0001389826,0.0003537667,0.000182994,0.007318974,0.00299633],"genre_scores_gemma":[0.1304706,0.0001190297,0.864866,0.0000875427,0.00008666054,0.0003559971,0.000450898,0.000453288,0.003110036],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.008945438,"threshold_uncertainty_score":0.04730856,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3087153552","doi":"10.1177/0265532220957298","title":"Hanyu Shuiping Kaoshi (HSK): A multi-level, multi-purpose proficiency test","year":2020,"lang":"en","type":"article","venue":"Language Testing","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":50,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"","keywords":"Language proficiency; Test (biology); Psychology; Argument (complex analysis); Language assessment; Scale (ratio); Mathematics education; Linguistics; Geography","authors":[{"name":"Yue Peng","is_ca":false},{"name":"Wei Yan","is_ca":true},{"name":"Liying Cheng","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.8227491867455936,"gpt":0.5099737195301289,"spread":0.3127754672154647,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007883831,0.0003635658,0.000728913,0.004107976,0.0002901108,0.0008506479,0.0007496743,0.0005461299,0.002374853],"category_scores_gemma":[0.01831223,0.0001553423,0.000503607,0.002219226,0.0005447093,0.001055391,0.0008729197,0.0006051452,0.0005317453],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009432597,"about_ca_system_score_gemma":0.008148988,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004707601,"about_ca_topic_score_gemma":0.00616891,"domain_scores_codex":[0.9964742,0.001218317,0.0006682827,0.0002312399,0.001332265,0.00007565485],"domain_scores_gemma":[0.9877635,0.005560908,0.001361621,0.0004230697,0.004363662,0.0005271061],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001663854,0.0001927761,0.01618507,0.006537811,0.0002786633,0.0001877622,0.0001759719,0.0002011855,0.001458754,0.001056403,0.0088497,0.9647096],"study_design_scores_gemma":[0.0008867993,0.005226997,0.4557226,0.01818327,0.004295383,0.005888433,0.001075979,0.004776957,0.02178539,0.005644336,0.4761869,0.0003269927],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"methods","genre_scores_codex":[0.2014338,0.6605176,0.04249473,0.0206019,0.003905125,0.005587209,0.006327467,0.001364362,0.05776779],"genre_scores_gemma":[0.6571354,0.2474284,0.067031,0.003393615,0.00122001,0.003545743,0.006821753,0.0001395268,0.01328453],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.007883831,"threshold_uncertainty_score":0.04169416,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2318200755","doi":"10.1177/0265532213509810","title":"Examining the impact of L2 proficiency and keyboarding skills on scores on TOEFL-iBT writing tasks","year":2013,"lang":"en","type":"article","venue":"Language Testing","topic":"Writing and Handwriting Education","field":"Social Sciences","cited_by":45,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"York University","funders":"","keywords":"Test of English as a Foreign Language; Psychology; Test (biology); Language proficiency; Task (project management); Context (archaeology); English language; Mathematics education","authors":[{"name":"Khaled Barkaoui","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.04433657734822507,"gpt":0.3557343872788868,"spread":0.3113978099306617,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002594875,0.0007747876,0.0005740676,0.00100547,0.0003168082,0.001042714,0.0005009238,0.0004478468,0.003148664],"category_scores_gemma":[0.01775092,0.0002147139,0.0008072724,0.0005511574,0.0006130955,0.0007461549,0.001022285,0.0007770308,0.0007648541],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003866562,"about_ca_system_score_gemma":0.0002775648,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004278461,"about_ca_topic_score_gemma":0.00519506,"domain_scores_codex":[0.9978167,0.0005957266,0.0002485995,0.0003454314,0.0007158386,0.0002777046],"domain_scores_gemma":[0.9737083,0.01461867,0.005825583,0.001337354,0.002348519,0.002161499],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0008459972,0.0005359622,0.9800465,0.00003695368,0.0001909351,0.0002692845,0.001171368,0.0002386138,0.003965766,0.00002139372,0.0001934804,0.01248382],"study_design_scores_gemma":[0.000006489735,0.0007449301,0.9979868,0.000003269268,0.00001720768,0.00007127903,0.0002306768,0.0001652532,0.000687492,0.000007853972,0.00007297423,0.000005749258],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9993669,0.00002689928,0.0000600972,0.00001109562,0.00000250757,0.000006077299,0.00007609324,0.000006527842,0.0004437536],"genre_scores_gemma":[0.9990152,0.00001755115,0.00009335104,0.00001073738,0.000004162146,0.000012792,0.0001759259,0.000005709448,0.000664519],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004278461,"threshold_uncertainty_score":0.01372319,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2751665091","doi":"10.1177/0265532217725776","title":"Developing and evaluating a computerized adaptive testing version of the Word Part Levels Test","year":2017,"lang":"en","type":"article","venue":"Language Testing","topic":"Second Language Acquisition and Learning","field":"Psychology","cited_by":45,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Western University","funders":"","keywords":"Affix; Test (biology); Vocabulary; Computerized adaptive testing; Natural language processing; Computer science; Strengths and weaknesses; Word (group theory); Vocabulary development; Psychology; Artificial intelligence; Linguistics; Psychometrics; Social psychology","authors":[{"name":"Atsushi Mizumoto","is_ca":false},{"name":"Yosuke Sasao","is_ca":false},{"name":"Stuart Webb","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1617082178408199,"gpt":0.3865890369439115,"spread":0.2248808191030916,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01608024,0.0007735039,0.0007038628,0.001609601,0.0005102998,0.001256546,0.001349946,0.0009056579,0.001399216],"category_scores_gemma":[0.03725248,0.0004235517,0.0007813913,0.001090033,0.0007604887,0.001498158,0.001282134,0.001183652,0.0006376651],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001336351,"about_ca_system_score_gemma":0.003239521,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007416,"about_ca_topic_score_gemma":0.009491154,"domain_scores_codex":[0.9883777,0.004914818,0.001767643,0.001162992,0.003386562,0.000390299],"domain_scores_gemma":[0.9744304,0.01279125,0.001561987,0.00140821,0.008850729,0.0009573869],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.002838269,0.01271909,0.5395364,0.000484187,0.0002866397,0.0005901551,0.005836051,0.005216554,0.016838,0.001023015,0.003366538,0.4112651],"study_design_scores_gemma":[0.001338274,0.02663551,0.9034678,0.000192812,0.0004755202,0.001903587,0.002600618,0.02110178,0.02901322,0.000924358,0.01212349,0.0002231711],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9786598,0.0002119476,0.01072461,0.0001527392,0.00008876573,0.006071549,0.0007079515,0.0002529371,0.003129761],"genre_scores_gemma":[0.8737003,0.000488284,0.1097217,0.0002997024,0.00006095397,0.009254976,0.003369177,0.0001129712,0.002991911],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01608024,"threshold_uncertainty_score":0.08504146,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2131222006","doi":"10.1177/0265532210364380","title":"Use of tree-based regression in the analyses of L2 reading test items","year":2010,"lang":"en","type":"article","venue":"Language Testing","topic":"Educational Technology and Assessment","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Alberta","funders":"","keywords":"Reading (process); Cognition; Psychology; Interpretation (philosophy); Cognitive psychology; Test (biology); Tree (set theory); Regression; Regression analysis; Natural language processing; Artificial intelligence; Computer science; Machine learning; Linguistics; Mathematics","authors":[{"name":"Lingyun Gao","is_ca":true},{"name":"W. Todd Rogers","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1065583235754579,"gpt":0.3862878005927542,"spread":0.2797294770172963,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04071942,0.002715764,0.001815071,0.004236252,0.0006670157,0.001988568,0.001418163,0.0007584753,0.002325297],"category_scores_gemma":[0.1662452,0.000658299,0.002386888,0.005610649,0.0005987318,0.002688258,0.001349382,0.002730869,0.00166329],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007155774,"about_ca_system_score_gemma":0.001230108,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006937715,"about_ca_topic_score_gemma":0.006278155,"domain_scores_codex":[0.9493062,0.04480658,0.001084514,0.002120448,0.002243699,0.0004385458],"domain_scores_gemma":[0.784122,0.1924188,0.007942911,0.007779472,0.007238776,0.000498054],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002189266,0.001038043,0.2203121,0.001130095,0.003148456,0.0006095589,0.005384243,0.1060365,0.01232934,0.01720651,0.005826845,0.624789],"study_design_scores_gemma":[0.0001726613,0.00183875,0.07227311,0.0002717504,0.0007602227,0.0005886003,0.001146354,0.8956475,0.006311596,0.01478573,0.005971241,0.0002325003],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1542628,0.0002376466,0.8382488,0.0001739249,0.000108935,0.0008729113,0.0007374007,0.002783603,0.002574039],"genre_scores_gemma":[0.5544535,0.0002437967,0.4403354,0.00009071834,0.00005335671,0.001299517,0.001198814,0.001056834,0.001268069],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.04071942,"threshold_uncertainty_score":0.2153475,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4387429753","doi":"10.1177/02655322231202947","title":"Our validity looks like justice. Does yours?","year":2023,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":27,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Lethbridge","funders":"","keywords":"Economic Justice; Oppression; Psychology; Licensure; Set (abstract data type); Social psychology; Mathematics education; Pedagogy; Law; Political science; Computer science","authors":[{"name":"Jennifer Randall","is_ca":false},{"name":"Mya Poe","is_ca":false},{"name":"David Slomp","is_ca":true},{"name":"María Elena Oliveri","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1345071057179136,"gpt":0.4132933080995925,"spread":0.2787862023816789,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1414046,0.0008038228,0.001649598,0.004897882,0.009235154,0.01317238,0.002094889,0.004590561,0.004955603],"category_scores_gemma":[0.4152649,0.0006694653,0.001356709,0.002530191,0.06089355,0.02036346,0.009169753,0.009881493,0.001884455],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006915985,"about_ca_system_score_gemma":0.01525861,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009504145,"about_ca_topic_score_gemma":0.007370447,"domain_scores_codex":[0.8181638,0.1047082,0.01075343,0.01388771,0.0485047,0.003982083],"domain_scores_gemma":[0.5677913,0.2039656,0.03484373,0.06526697,0.1184091,0.009723244],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002285718,0.0003241505,0.04270772,0.001142408,0.0004475033,0.0002711135,0.0584848,0.000472072,0.00173908,0.6345168,0.05859088,0.201075],"study_design_scores_gemma":[0.00007377678,0.0003683872,0.0180102,0.002645706,0.0002160036,0.0004913288,0.033465,0.001420263,0.002113769,0.7494276,0.1915257,0.0002422938],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.05509285,0.004960268,0.07209195,0.6691323,0.01054317,0.0005949302,0.0005216009,0.0004072149,0.1866557],"genre_scores_gemma":[0.8707526,0.001652109,0.0331813,0.07887679,0.003154908,0.0006152446,0.0001900357,0.0004189205,0.01115794],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8585954,"threshold_uncertainty_score":0.7478278,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2917835789","doi":"10.1177/0265532219828252","title":"The Test of English for International Communication (TOEIC <sup>®</sup> )","year":2019,"lang":"en","type":"article","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":22,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Queen's University","funders":"American Psychological Association; Educational Testing Service","keywords":"TOEIC; Psychology; Test (biology); Linguistics; International communication; Language proficiency; Language assessment; Mathematics education; Communication; Philosophy; Reading (process)","authors":[{"name":"Gwan-Hyeok Im","is_ca":true},{"name":"Liying Cheng","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.02662448091769125,"gpt":0.2592560255326446,"spread":0.2326315446149533,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001613151,0.001122903,0.0005404858,0.002266488,0.000584983,0.001150307,0.0008042047,0.001096357,0.02668701],"category_scores_gemma":[0.01081583,0.0001733513,0.0006431573,0.000704015,0.0006626694,0.001575217,0.001631766,0.0008461839,0.0148067],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003277424,"about_ca_system_score_gemma":0.001325406,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003525054,"about_ca_topic_score_gemma":0.003449828,"domain_scores_codex":[0.9980291,0.0004216723,0.0002450107,0.0001533894,0.0009356521,0.0002152527],"domain_scores_gemma":[0.995529,0.001374177,0.0004166894,0.0002971539,0.001764629,0.0006184389],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001877107,0.001144678,0.1533696,0.0005434066,0.0001164225,0.00204916,0.001674865,0.0008399543,0.0275058,0.006204806,0.1947449,0.6099294],"study_design_scores_gemma":[0.0003174195,0.003219717,0.6530762,0.0005114068,0.0001208981,0.01058066,0.002796044,0.003056915,0.05279798,0.005450765,0.267825,0.0002469778],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.425234,0.002361284,0.03000061,0.003756622,0.003363335,0.002579929,0.03951098,0.003376534,0.4898167],"genre_scores_gemma":[0.6584084,0.001657641,0.044694,0.003115316,0.0004126492,0.004603006,0.03265594,0.001304028,0.253149],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02668701,"threshold_uncertainty_score":0.08927691,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2097625210","doi":"10.1177/0265532211404195","title":"Test review: ACCESS for ELLs <sup>®</sup>","year":2011,"lang":"en","type":"article","venue":"Language Testing","topic":"Digital Accessibility for Disabilities","field":"Social Sciences","cited_by":16,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Carleton University","funders":"","keywords":"Ell; Psychology; Test (biology); Language proficiency; Mathematics education; Teaching method; Geology","authors":[{"name":"Janna Fox","is_ca":true},{"name":"Shelley Fairbairn","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1337333077180752,"gpt":0.3811891609412508,"spread":0.2474558532231756,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008904257,0.001093938,0.002257954,0.004807801,0.0009007775,0.00259656,0.002525527,0.004017615,0.03004872],"category_scores_gemma":[0.05556104,0.0003270379,0.001467699,0.003879894,0.001178419,0.002425928,0.001683206,0.001702741,0.009522787],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003127009,"about_ca_system_score_gemma":0.01113778,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01287857,"about_ca_topic_score_gemma":0.02879425,"domain_scores_codex":[0.9929445,0.002518984,0.001380812,0.0002921138,0.002514786,0.0003489136],"domain_scores_gemma":[0.9455424,0.02070571,0.005420247,0.0008102643,0.02489896,0.002622413],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003812473,0.00008989713,0.0006822021,0.02383716,0.0001691613,0.0001814362,0.00008837107,0.00006076592,0.0004976707,0.0008145637,0.6007603,0.3724371],"study_design_scores_gemma":[0.0003318097,0.0006418225,0.00772302,0.05290999,0.001178428,0.0007828777,0.0003127724,0.0001309024,0.001203396,0.001050576,0.9336714,0.00006294154],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.003509116,0.8056568,0.001562878,0.09603358,0.04300225,0.001409912,0.00451857,0.0003201432,0.04398683],"genre_scores_gemma":[0.03543705,0.7901912,0.00475729,0.06598925,0.01749501,0.002111738,0.008913721,0.0002619448,0.07484279],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.03004872,"threshold_uncertainty_score":0.100523,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4383069305","doi":"10.1177/02655322231179134","title":"Fairness of using different English accents: The effect of shared L1s in listening tasks of the Duolingo English test","year":2023,"lang":"en","type":"article","venue":"Language Testing","topic":"Phonetics and Phonology Research","field":"Psychology","cited_by":15,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Brock University","funders":"","keywords":"Active listening; Psychology; Stress (linguistics); Test (biology); Interlanguage; Vocabulary; Dictation; Linguistics; Task (project management); Hindi; Communication","authors":[{"name":"Okim Kang","is_ca":false},{"name":"Xun Yan","is_ca":false},{"name":"Maria Kostromitina","is_ca":false},{"name":"Ron I. Thomson","is_ca":true},{"name":"Talia Isaacs","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03822191359303156,"gpt":0.3483421313789877,"spread":0.3101202177859562,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05847613,0.0005535468,0.0006600661,0.0006298637,0.001085292,0.001914751,0.0007575943,0.0007678882,0.001182708],"category_scores_gemma":[0.1640635,0.000420373,0.0006202831,0.0003171373,0.002398172,0.001862976,0.003694753,0.001266797,0.000337652],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006339923,"about_ca_system_score_gemma":0.0006180952,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001090895,"about_ca_topic_score_gemma":0.002264256,"domain_scores_codex":[0.95171,0.03043661,0.003753136,0.004363387,0.008815587,0.0009212358],"domain_scores_gemma":[0.6981454,0.240359,0.02795635,0.02117159,0.008810516,0.003557123],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.01761842,0.002189803,0.7263381,0.0003865261,0.0008199843,0.000557832,0.02989757,0.001740327,0.0844661,0.00154769,0.0004234676,0.1340142],"study_design_scores_gemma":[0.0002234212,0.005909177,0.9531065,0.0001205352,0.0002940425,0.0007106285,0.004002206,0.002249994,0.03041326,0.001486908,0.001336811,0.0001463486],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9928139,0.0001444125,0.004023078,0.0001274018,0.00003864456,0.00008621126,0.0000220578,0.00002164722,0.002722617],"genre_scores_gemma":[0.9971232,0.00003298772,0.002153491,0.0001489828,0.00002796179,0.00004952927,0.00002416686,0.00002279969,0.0004169266],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.05847613,"threshold_uncertainty_score":0.3092551,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2782384206","doi":"10.1177/0265532217750692","title":"Examining sources of variability in repeaters’ L2 writing scores: The case of the PTE Academic writing section","year":2018,"lang":"en","type":"article","venue":"Language Testing","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"York University","funders":"","keywords":"Test (biology); Psychology; Context (archaeology); Boston Naming Test; Language assessment; Language proficiency; Mathematics education; Cognition","authors":[{"name":"Khaled Barkaoui","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.3740090240133829,"gpt":0.4424069973241175,"spread":0.0683979733107346,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01680917,0.0005903814,0.0009403958,0.002077518,0.0007889246,0.001449179,0.001232195,0.0009154358,0.0008750019],"category_scores_gemma":[0.08644085,0.0004298863,0.001207516,0.001830958,0.0008822128,0.001172158,0.002051326,0.001258612,0.0003054514],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007867691,"about_ca_system_score_gemma":0.0007109115,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009939506,"about_ca_topic_score_gemma":0.01300357,"domain_scores_codex":[0.9786649,0.009484821,0.001519954,0.004085725,0.005554582,0.0006899992],"domain_scores_gemma":[0.882834,0.06778924,0.02032795,0.01847431,0.009355581,0.001218942],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001650633,0.00008151137,0.9730865,0.00004991613,0.0004904555,0.0002261814,0.006504957,0.0004934868,0.001100683,0.0002335684,0.0002576248,0.01730991],"study_design_scores_gemma":[0.000004315327,0.0001648292,0.9944717,0.00002485887,0.00006522344,0.00046772,0.001316674,0.001920923,0.000724516,0.0002548628,0.0005541747,0.00003021488],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.991251,0.0003736107,0.006684916,0.0001639738,0.00002614572,0.000063297,0.0003714664,0.00005441561,0.001011177],"genre_scores_gemma":[0.9973416,0.00005012375,0.001704032,0.00003121624,0.00001686343,0.00004417276,0.000371491,0.00002976491,0.0004107461],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01680917,"threshold_uncertainty_score":0.08889645,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4407931412","doi":"10.1177/02655322251319284","title":"The relationship between English language proficiency test scores and academic achievement: A longitudinal study of two tests","year":2025,"lang":"en","type":"article","venue":"Language Testing","topic":"Higher Education Learning Practices","field":"Social Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"York University","funders":"","keywords":"Psychology; Language proficiency; Test (biology); Language assessment; Achievement test; Academic achievement; Mathematics education; English language; Test of English as a Foreign Language; Standardized test","authors":[{"name":"Khaled Barkaoui","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.1073036299005186,"gpt":0.4529902981757598,"spread":0.3456866682752412,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003607478,0.0003226394,0.0003787478,0.001176498,0.001590806,0.00120793,0.0006249201,0.0006640399,0.0009654352],"category_scores_gemma":[0.00618503,0.0003496382,0.0006463816,0.0008280507,0.0005909179,0.001064584,0.001323449,0.001910691,0.0004568044],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009516902,"about_ca_system_score_gemma":0.001494855,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0320573,"about_ca_topic_score_gemma":0.0433431,"domain_scores_codex":[0.9988966,0.0003235363,0.00008000054,0.0001901206,0.0002492935,0.0002603764],"domain_scores_gemma":[0.9942625,0.0007995369,0.00156428,0.0004889593,0.001219451,0.001665159],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00009697174,0.0003605229,0.9965993,0.000002926676,0.00004129541,0.00005618694,0.0008394063,0.00002212848,0.0001843897,0.00002998337,0.00007802354,0.001688904],"study_design_scores_gemma":[0.000005235174,0.0004228163,0.9977024,0.0000072039,0.00002344721,0.00008078521,0.001175869,0.0001274218,0.0001486885,0.00002794593,0.0002691392,0.000008972283],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9995515,0.00005118058,0.0000523905,0.00003801438,0.000004100916,0.000007143208,0.00009026697,0.000001938664,0.0002035743],"genre_scores_gemma":[0.9990771,0.0000403565,0.00008264254,0.000021738,0.000003211619,0.00001725897,0.0002352027,0.000002029534,0.0005205034],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0320573,"threshold_uncertainty_score":0.06374145,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3039959558","doi":"10.1177/0265532220930348","title":"Change in home language environment and English literacy achievement over time: A multi-group latent growth curve modeling investigation","year":2020,"lang":"en","type":"article","venue":"Language Testing","topic":"Multilingual Education and Policy","field":"Social Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Toronto","funders":"","keywords":"Latent growth modeling; Literacy; Psychology; Longitudinal study; Home language; Competence (human resources); Population; Academic achievement; Mathematics education; Immigration; Developmental psychology; Pedagogy; Social psychology; Demography; Sociology; Geography; Medicine","authors":[{"name":"Christine Barron","is_ca":true},{"name":"Jeanne Sinclair","is_ca":false},{"name":"Eunice Eunhee Jang","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.08900675997372587,"gpt":0.3676722898153619,"spread":0.278665529841636,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01480091,0.001228836,0.001354693,0.002320973,0.001674204,0.002450961,0.002585475,0.001338283,0.003474491],"category_scores_gemma":[0.01958636,0.0005486811,0.003373406,0.002736117,0.001396559,0.001483276,0.002990952,0.002644012,0.0008432713],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003007442,"about_ca_system_score_gemma":0.004235401,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.1867547,"about_ca_topic_score_gemma":0.1068935,"domain_scores_codex":[0.9943381,0.003421572,0.0001917666,0.0009634132,0.0004750197,0.000610163],"domain_scores_gemma":[0.986954,0.007510118,0.001709213,0.001842228,0.00128314,0.0007012852],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0005443046,0.0006000175,0.9584889,0.0000731117,0.0006939907,0.0002551494,0.005252766,0.01292002,0.0004707954,0.002607469,0.001299664,0.01679363],"study_design_scores_gemma":[0.0001013777,0.0007735501,0.5940625,0.0001703255,0.0005384912,0.000351403,0.0108237,0.3835873,0.0005651512,0.00471598,0.004175321,0.0001348706],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9857897,0.000195983,0.01174076,0.0004796044,0.00002370457,0.0001555278,0.0008721963,0.0001054877,0.0006370753],"genre_scores_gemma":[0.9916382,0.0001118012,0.004944407,0.00003548658,0.000008064278,0.000224932,0.001338144,0.00003498659,0.001663984],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1867547,"threshold_uncertainty_score":0.3713354,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W3214027731","doi":"10.1177/02655322211052680","title":"Investigating and optimizing score dependability of a local ITA speaking test across language groups: A generalizability theory approach","year":2021,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":10,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Generalizability theory; Dependability; Language proficiency; Psychology; Variance (accounting); Test (biology); Construct (python library); Formative assessment; Computer science; Mathematics education; Developmental psychology; Accounting","authors":[{"name":"Ji-young Shin","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.05086898553121055,"gpt":0.3497865942429446,"spread":0.2989176087117341,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.08313917,0.001532721,0.001400314,0.003990026,0.0008686305,0.00257611,0.001762122,0.001037056,0.00165519],"category_scores_gemma":[0.2476727,0.0006271307,0.002676324,0.002965448,0.002893576,0.003257561,0.003501665,0.00179169,0.0002912901],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00180392,"about_ca_system_score_gemma":0.001927069,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004384214,"about_ca_topic_score_gemma":0.003646592,"domain_scores_codex":[0.9418617,0.04227727,0.002331218,0.005780945,0.006960205,0.0007885899],"domain_scores_gemma":[0.7353418,0.2129482,0.01093709,0.02520711,0.01471877,0.0008469287],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0009344204,0.0006063585,0.6727764,0.0003843546,0.002189075,0.0001947402,0.008351212,0.0154744,0.006535386,0.009704689,0.0005476653,0.2823014],"study_design_scores_gemma":[0.0002313248,0.005896743,0.8465566,0.0002166998,0.001621534,0.0003817634,0.004894767,0.09727561,0.01561534,0.02442326,0.002732716,0.0001535674],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6361603,0.0004198766,0.3526607,0.0006285344,0.00005691012,0.001110096,0.0002024846,0.0004336982,0.008327438],"genre_scores_gemma":[0.9507976,0.00006967164,0.0478145,0.00008031257,0.00002528979,0.0005643389,0.0001573639,0.0000704249,0.0004204357],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.08313917,"threshold_uncertainty_score":0.4396873,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4414209357","doi":"10.1177/02655322251348685","title":"Advancing language assessment for teaching and learning in the era of the artificial intelligence (AI) revolution: Promises and challenges","year":2025,"lang":"en","type":"article","venue":"Language Testing","topic":"Second Language Learning and Teaching","field":"Arts and Humanities","cited_by":10,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"Japan Society for the Promotion of Science; Waseda University","keywords":"Language assessment; Language proficiency; Language acquisition; Applications of artificial intelligence; Assessment for learning; Teaching method","authors":[{"name":"Eunice Eunhee Jang","is_ca":true},{"name":"Yasuyo Sawaki","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0365263602611972,"gpt":0.3059178929668351,"spread":0.2693915327056379,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0604332,0.001100538,0.001660472,0.003185197,0.002531937,0.01572352,0.004402491,0.006751264,0.01120257],"category_scores_gemma":[0.140077,0.0002989584,0.0004747821,0.001962687,0.01246689,0.03060929,0.011512,0.01056729,0.003034393],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006519475,"about_ca_system_score_gemma":0.02125966,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007479151,"about_ca_topic_score_gemma":0.00806505,"domain_scores_codex":[0.9674701,0.01989641,0.001298442,0.00124607,0.008777518,0.001311458],"domain_scores_gemma":[0.7965822,0.1352032,0.006129441,0.007563136,0.03781469,0.01670727],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002563958,0.0006985995,0.01041673,0.0008364965,0.00003727399,0.0001266425,0.003653684,0.001268962,0.001109696,0.1569981,0.03932201,0.7852754],"study_design_scores_gemma":[0.00008860134,0.0006662869,0.007008088,0.002417104,0.00002991052,0.0003990383,0.01094088,0.009736213,0.003162809,0.7752033,0.1901772,0.0001705552],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"review","genre_scores_codex":[0.03415008,0.05911489,0.1031336,0.7349624,0.004024211,0.0001685056,0.0003299619,0.001883289,0.06223312],"genre_scores_gemma":[0.7340036,0.03911021,0.164066,0.03933423,0.006668656,0.0003328106,0.000452124,0.0005935305,0.01543888],"genre_candidate":"review","genre_consensus":null,"teacher_disagreement_score":0.0604332,"threshold_uncertainty_score":0.3196051,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2337150116","doi":"10.1177/0265532214559115","title":"Design in four diagnostic language assessments","year":2015,"lang":"en","type":"article","venue":"Language Testing","topic":"Educational and Psychological Assessments","field":"Psychology","cited_by":9,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Toronto","funders":"","keywords":"Language assessment; Language proficiency; Psychology; Educational assessment; Process (computing); Intervention (counseling); Educational research; Mathematics education; Management science; Computer science; Pedagogy","authors":[{"name":"Alister Cumming","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2169898390405153,"gpt":0.4566473782595891,"spread":0.2396575392190738,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04240267,0.0009022532,0.0007321286,0.004225131,0.002880648,0.006577973,0.003126456,0.002231469,0.007169635],"category_scores_gemma":[0.09704426,0.0009262599,0.0009327391,0.002483279,0.004181888,0.003782096,0.008320687,0.002305858,0.002214636],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006543671,"about_ca_system_score_gemma":0.01134942,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001704589,"about_ca_topic_score_gemma":0.002881874,"domain_scores_codex":[0.9487635,0.03257975,0.006099351,0.004137253,0.00663344,0.001786844],"domain_scores_gemma":[0.9171156,0.04341727,0.006101175,0.01032144,0.01940824,0.003636196],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.002408999,0.002323169,0.05155953,0.003210473,0.0001330811,0.0008577225,0.0805931,0.003390515,0.008093774,0.1151241,0.01237813,0.7199275],"study_design_scores_gemma":[0.002325702,0.005836597,0.05971565,0.004425993,0.0006106213,0.002487499,0.0804157,0.01607157,0.0347132,0.1686709,0.6241461,0.0005804698],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.394625,0.003220085,0.4009633,0.006334351,0.001475645,0.03884285,0.001441421,0.002413718,0.1506836],"genre_scores_gemma":[0.4884155,0.0007269115,0.4682192,0.001840254,0.00008664049,0.02273213,0.0007693658,0.000248319,0.01696162],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.04240267,"threshold_uncertainty_score":0.2242494,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2168715918","doi":"10.1177/02655322090260020602","title":"Test review: The Versant Spanish <sup>TM</sup> Test","year":2009,"lang":"en","type":"article","venue":"Language Testing","topic":"Neurobiology of Language and Bilingualism","field":"Neuroscience","cited_by":8,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Carleton University","funders":"","keywords":"Test (biology); Psychology; Geology","authors":[{"name":"Janna Fox","is_ca":true},{"name":"Wendy Fraser","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03386165576883628,"gpt":0.2910663115460739,"spread":0.2572046557772376,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002463794,0.0007373766,0.001191754,0.001361285,0.0004766186,0.001076013,0.001925807,0.002113424,0.01560763],"category_scores_gemma":[0.01008132,0.0001840889,0.0004685198,0.001109655,0.0007596905,0.0009967093,0.0006631064,0.001163921,0.009289597],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001286155,"about_ca_system_score_gemma":0.00331052,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00697942,"about_ca_topic_score_gemma":0.01730384,"domain_scores_codex":[0.9988005,0.0003370694,0.000149937,0.0001168897,0.0005280547,0.00006755192],"domain_scores_gemma":[0.9925505,0.00194043,0.0004243231,0.0003435418,0.004209574,0.0005315853],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004830442,0.0001256557,0.002446895,0.001443317,0.00007768178,0.0008513773,0.00003094462,0.00009008203,0.001411365,0.001109517,0.4261851,0.5657451],"study_design_scores_gemma":[0.0001643013,0.0005448613,0.0102214,0.001372246,0.0001577745,0.004133295,0.00008389581,0.0001653089,0.002355871,0.001218302,0.979548,0.00003478657],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"empirical","genre_scores_codex":[0.03266427,0.5737682,0.008308944,0.1198631,0.06584161,0.0006714221,0.006571662,0.001727678,0.1905832],"genre_scores_gemma":[0.1655442,0.4158945,0.01366354,0.08308638,0.03235475,0.0006834267,0.01389367,0.001572191,0.2733074],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01560763,"threshold_uncertainty_score":0.05221272,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4251721468","doi":"10.1177/0265532220925448","title":"Understanding writing quality change: A longitudinal study of repeaters of a high-stakes standardized English proficiency test","year":2020,"lang":"en","type":"article","venue":"Language Testing","topic":"Writing and Handwriting Education","field":"Social Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"","funders":"","keywords":"Language proficiency; Psychology; Test (biology); Sophistication; Linguistics; Mathematics education","authors":[{"name":"You-Min Lin","is_ca":false},{"name":"Michelle Y. Chen","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.4041793967710166,"gpt":0.4111835522633015,"spread":0.007004155492284891,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001772558,0.0003104826,0.0004894471,0.0009956155,0.0008225604,0.0007769205,0.0006696213,0.0008115863,0.0008167771],"category_scores_gemma":[0.008109648,0.0003281676,0.0005305953,0.0006929956,0.0004082769,0.0007759631,0.0006273013,0.0009766462,0.0004680221],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005514867,"about_ca_system_score_gemma":0.0004344385,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01896972,"about_ca_topic_score_gemma":0.02083797,"domain_scores_codex":[0.9989138,0.0002186481,0.0000995997,0.0001801501,0.0004119833,0.0001758825],"domain_scores_gemma":[0.9937552,0.0008144168,0.002485103,0.0005324286,0.001644813,0.0007680209],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00007691726,0.0001943209,0.9940573,0.000006344445,0.00003903639,0.0001603362,0.001693944,0.00002774115,0.0004612088,0.000009713145,0.00006621791,0.003206789],"study_design_scores_gemma":[0.00000198218,0.0002879178,0.9984547,0.000002839126,0.000009066498,0.0001885287,0.0007001489,0.00009094222,0.0001181114,0.000007958578,0.0001323055,0.000005528087],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9996734,0.00004487671,0.00005598756,0.00002000555,0.000002046551,0.000007708416,0.00006059833,0.000002364795,0.0001330857],"genre_scores_gemma":[0.9993008,0.00002569825,0.00007934061,0.00001217192,0.000002639247,0.000009019112,0.0001659085,0.000002683144,0.0004017013],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01896972,"threshold_uncertainty_score":0.03771859,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W1979398480","doi":"10.1177/0265532214560799","title":"A prototype of a receptive lexical test for a polysynthetic heritage language: The case of Inuttitut in Labrador","year":2014,"lang":"en","type":"article","venue":"Language Testing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"The Scarborough Hospital; University of Toronto; Carleton University","funders":"","keywords":"Heritage language; Linguistics; Vocabulary; Psychology; Test (biology); Comprehension; Noun; Language proficiency; First language; Natural language processing; Computer science; Mathematics education","authors":[{"name":"Marina Sherkina-Lieber","is_ca":true},{"name":"Rena Helms‐Park","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01657557252071323,"gpt":0.2924684211711534,"spread":0.2758928486504402,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001786797,0.000557159,0.0003320172,0.0009145301,0.001237666,0.0009160139,0.001368707,0.0008671948,0.003045206],"category_scores_gemma":[0.002820899,0.0002486055,0.0002425395,0.0004898448,0.001349194,0.0008349089,0.001114497,0.0008388538,0.00118376],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007900199,"about_ca_system_score_gemma":0.001944705,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01756926,"about_ca_topic_score_gemma":0.05738471,"domain_scores_codex":[0.9990737,0.0004099875,0.00008657813,0.0001366588,0.000107359,0.0001856998],"domain_scores_gemma":[0.9989163,0.0003780137,0.0001078801,0.0001548774,0.0002361238,0.0002068478],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0009540868,0.004352829,0.4051856,0.0005323555,0.00003182844,0.05747019,0.1050179,0.00114885,0.0899139,0.004290867,0.005081096,0.3260205],"study_design_scores_gemma":[0.000270506,0.008038996,0.5610843,0.0006933736,0.0001155838,0.0978569,0.14855,0.007194425,0.07659984,0.004925144,0.09430142,0.000369617],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9823238,0.0001388267,0.007040344,0.0005131693,0.00004365841,0.000332472,0.0001817493,0.0002307044,0.0091953],"genre_scores_gemma":[0.9470077,0.0002215172,0.04564352,0.0003505608,0.00001841409,0.0003804668,0.0002632988,0.00008933125,0.006025229],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9824308,"threshold_uncertainty_score":0.03493398,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W2162394022","doi":"10.1177/0265532212469178","title":"Differential importance of language components in determining secondary school students’ Chinese reading literacy performance","year":2013,"lang":"en","type":"article","venue":"Language Testing","topic":"Reading and Literacy Development","field":"Psychology","cited_by":6,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Dictation; Psychology; Reading (process); Copying; Reading comprehension; Literacy; Mathematics education; Chinese characters; Linguistics; Pedagogy; Computer science; Artificial intelligence","authors":[{"name":"Che Kan Leong","is_ca":true},{"name":"Man Koon Ho","is_ca":false},{"name":"Jianfang Chang","is_ca":false},{"name":"Kit‐Tai Hau","is_ca":false}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01812018530720183,"gpt":0.3170660889351399,"spread":0.298945903627938,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006790824,0.0004385083,0.0003074897,0.001368091,0.0003722688,0.0007151695,0.0002330743,0.0002591197,0.001379534],"category_scores_gemma":[0.002591591,0.0002094379,0.0002925985,0.000856646,0.0005258727,0.0003791577,0.0005324883,0.0003063433,0.0004034413],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003608872,"about_ca_system_score_gemma":0.000507594,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01442404,"about_ca_topic_score_gemma":0.02383716,"domain_scores_codex":[0.9994478,0.00009321092,0.00009418823,0.0001216114,0.0001302847,0.0001128421],"domain_scores_gemma":[0.9982094,0.0004207637,0.0005813711,0.0001122441,0.0003297741,0.0003465138],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00004234772,0.00003505239,0.9940037,0.00001031654,0.00001297667,0.0001023091,0.001088094,0.00001754003,0.001412916,0.00001596767,0.00002111405,0.003237674],"study_design_scores_gemma":[9.509521e-7,0.00004217588,0.9993067,0.000001246085,0.000004468772,0.00004084109,0.0003600137,0.0000400161,0.000157042,0.000007419055,0.00003770011,0.0000014844],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9997149,0.00001321932,0.00001767539,0.000003871804,7.437663e-7,0.000003734006,0.00002807374,0.00000146432,0.0002163331],"genre_scores_gemma":[0.9997008,0.00001332868,0.00004022146,0.000003684456,8.081782e-7,0.000004426587,0.00007036234,9.9218e-7,0.0001653628],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01442404,"threshold_uncertainty_score":0.02868015,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4399235559","doi":"10.1177/02655322241249754","title":"A scoping review of research on second language test preparation","year":2024,"lang":"en","type":"review","venue":"Language Testing","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":4,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"Université de Sherbrooke; Western University","funders":"","keywords":"Psychology; Test (biology); Language assessment; Language proficiency; Linguistics; Mathematics education","authors":[{"name":"Shanshan He","is_ca":true},{"name":"Anne-Marie Sénécal","is_ca":true},{"name":"Laura Stansfield","is_ca":true},{"name":"Ruslan Suvorov","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.2975427459650408,"gpt":0.5064660563792833,"spread":0.2089233104142425,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01513831,0.001642628,0.003379622,0.02057996,0.001446123,0.003875625,0.002123595,0.002591395,0.006144745],"category_scores_gemma":[0.07120959,0.001115859,0.004001724,0.02352789,0.001557179,0.003956672,0.002406575,0.00214661,0.001254846],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003960383,"about_ca_system_score_gemma":0.02597819,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009294444,"about_ca_topic_score_gemma":0.02332965,"domain_scores_codex":[0.9896711,0.002980958,0.003950057,0.0007174691,0.002413673,0.0002668409],"domain_scores_gemma":[0.9360244,0.04908827,0.005240827,0.00114128,0.0080217,0.0004836186],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","study_design_scores_codex":[0.00006536722,0.00003965463,0.0005737214,0.6183488,0.0006399336,0.0001969311,0.001156662,0.0001899225,0.0003553187,0.002418108,0.01096937,0.3650461],"study_design_scores_gemma":[0.00001381195,0.00005164253,0.001353797,0.8831227,0.002077254,0.0003470455,0.0006666887,0.00004379878,0.0001978392,0.0009581991,0.1111456,0.0000215546],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0003034035,0.9972104,0.0003312869,0.0005368813,0.0002637242,0.0001556663,0.0001502954,0.00001022752,0.001038198],"genre_scores_gemma":[0.001797186,0.9967065,0.0006047163,0.0002632626,0.00006814412,0.0002355154,0.0001535798,0.000005350446,0.0001656991],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.02057996,"threshold_uncertainty_score":0.08006001,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4381740816","doi":"10.1177/02655322231179128","title":"Rethinking student placement to enhance efficiency and student agency","year":2023,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":true},"ca_institutions":"Université de Montréal; Carleton University; University of Ottawa","funders":"University of Ottawa","keywords":"Agency (philosophy); Context (archaeology); Psychology; Process (computing); Test (biology); Mathematics education; Language proficiency; Pedagogy; Computer science; Sociology","authors":[{"name":"Beverly Baker","is_ca":true},{"name":"Angel Arias","is_ca":true},{"name":"Louis-David Bibeau","is_ca":true},{"name":"Coral Yiwei Qin","is_ca":true},{"name":"Margret Norenberg","is_ca":true},{"name":"Jennifer St-John","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.0478632348052326,"gpt":0.4160427232575863,"spread":0.3681794884523537,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06085993,0.001127809,0.001199527,0.002030472,0.002281797,0.009363875,0.002989932,0.001283933,0.004205599],"category_scores_gemma":[0.1869035,0.000583939,0.0008571433,0.001221952,0.002753547,0.004686075,0.005997197,0.003083749,0.002235246],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002821502,"about_ca_system_score_gemma":0.005627938,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001347309,"about_ca_topic_score_gemma":0.003621288,"domain_scores_codex":[0.9360338,0.04257496,0.003972036,0.002247844,0.01283562,0.002335692],"domain_scores_gemma":[0.8475259,0.08954683,0.007649707,0.0217089,0.02574314,0.007825507],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004195262,0.004196741,0.03817175,0.0005299443,0.00005951537,0.0002348143,0.05844525,0.001648725,0.01443919,0.004054585,0.007474994,0.870325],"study_design_scores_gemma":[0.0008399876,0.02330497,0.3756803,0.002428996,0.0002603844,0.00156057,0.1915412,0.04347396,0.106911,0.04801128,0.2045965,0.001390902],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8548789,0.0001903927,0.1084944,0.003935029,0.0006260409,0.003991201,0.0001034267,0.002326856,0.0254537],"genre_scores_gemma":[0.807312,0.0002232055,0.1821231,0.0004379776,0.00009896883,0.001828462,0.0001159599,0.0003297013,0.007530714],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.06085993,"threshold_uncertainty_score":0.3218619,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4404521715","doi":"10.1177/02655322241291764","title":"Review of the Canadian English Language Proficiency Index Program (CELPIP)","year":2024,"lang":"en","type":"article","venue":"Language Testing","topic":"Multilingual Education and Policy","field":"Social Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true},"ca_institutions":"University of Ottawa","funders":"","keywords":"Psychology; Language proficiency; Index (typography); Language assessment; Linguistics; Mathematics education; Computer science; Philosophy","authors":[{"name":"Coral Yiwei Qin","is_ca":true},{"name":"Beverly Baker","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.06094631709999317,"gpt":0.4570166644197495,"spread":0.3960703473197564,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01124023,0.001329065,0.002644819,0.01413054,0.001641176,0.002882551,0.004203777,0.001484925,0.004184624],"category_scores_gemma":[0.0482739,0.0007624007,0.001550338,0.0218222,0.002040531,0.001657416,0.00143364,0.002196798,0.0009144878],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.02296906,"about_ca_system_score_gemma":0.1158589,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.7186103,"about_ca_topic_score_gemma":0.7698047,"domain_scores_codex":[0.9932414,0.001160917,0.001244403,0.0005156099,0.003504454,0.0003331156],"domain_scores_gemma":[0.9639786,0.009873432,0.002005806,0.0004564847,0.02250762,0.001177896],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001093469,0.00004132051,0.001717048,0.1245995,0.0004188359,0.0001288285,0.0006861627,0.0002853578,0.0001725546,0.005725892,0.1924502,0.673665],"study_design_scores_gemma":[0.00002959477,0.00005477446,0.01258362,0.1708757,0.0009104447,0.0002538004,0.0004675095,0.00008275242,0.0001671111,0.0005443992,0.8139516,0.00007876923],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0002092657,0.9937338,0.0001273285,0.00236673,0.0005809011,0.00005752441,0.0007821648,0.00001125667,0.002130951],"genre_scores_gemma":[0.002631494,0.9942318,0.0005083142,0.001256082,0.0001781688,0.00009108759,0.0006587278,0.00001029759,0.0004339839],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.2813897,"threshold_uncertainty_score":0.5660937,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4412969591","doi":"10.1177/02655322251348956","title":"Investigating construct representativeness and linguistic equity of automated oral reading fluency assessment with prosody","year":2025,"lang":"en","type":"article","venue":"Language Testing","topic":"Reading and Literacy Development","field":"Psychology","cited_by":3,"is_retracted":false,"has_abstract":true,"routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false},"ca_institutions":"Institute for Christian Studies","funders":"Iran Science Elites Federation; University of Toronto","keywords":"Prosody; Fluency; Psychology; Ell; Natural language processing; Linguistics; Reading comprehension; Reading (process); Cognitive psychology; Computer science; Artificial intelligence; Mathematics education; Speech recognition; Teaching method; Vocabulary development","authors":[{"name":"Liam Hannah","is_ca":true},{"name":"Eunice Eunhee Jang","is_ca":true},{"name":"Meng‐Hsun Lee","is_ca":true},{"name":"Bruce W. Russell","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.03767559474110727,"gpt":0.4197115117810315,"spread":0.3820359170399242,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04922393,0.0004996574,0.0005139352,0.002359252,0.0005665605,0.002038297,0.0007004577,0.0007746584,0.0009193001],"category_scores_gemma":[0.1266352,0.0003418778,0.0009727496,0.001156365,0.001706484,0.001848604,0.003654448,0.0007623974,0.000271189],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006286842,"about_ca_system_score_gemma":0.0006914506,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001308608,"about_ca_topic_score_gemma":0.002516886,"domain_scores_codex":[0.9681037,0.01937824,0.002110931,0.003604987,0.006220652,0.0005814502],"domain_scores_gemma":[0.8593172,0.1038139,0.01161025,0.01145614,0.01273185,0.001070573],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000521718,0.0002715379,0.9383464,0.00008064509,0.0004402846,0.00005709891,0.004310286,0.001142688,0.002505312,0.00054202,0.0001286994,0.05165318],"study_design_scores_gemma":[0.0000492138,0.001577709,0.9746184,0.00007102369,0.0002728486,0.0003153568,0.002158748,0.01407197,0.004425126,0.001633821,0.0007606396,0.00004516495],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9832049,0.0001433459,0.01354999,0.00006122079,0.00001534769,0.0001729258,0.00006159941,0.00004469272,0.002745894],"genre_scores_gemma":[0.9954163,0.00002919574,0.004059135,0.00002254476,0.00001032655,0.000169752,0.00008431254,0.00001394847,0.0001945344],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.04922393,"threshold_uncertainty_score":0.2603241,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null},{"id":"W4412676320","doi":"10.1177/02655322251359320","title":"Book review: Innovation in Learning-Oriented Language Assessment ChongS.ReindersH. (Eds.), Innovation in Learning-Oriented Language Assessment. Palgrave MacMillan, 2023. 333 pp. ISBN 978-3-031-18949-4 (hbk) US$169.99 ISBN 978-3-031-18952-4 (sbk) US$169.99 ISBN 978-3-031-18950-0 (ebk) US$129.99","year":2025,"lang":"en","type":"article","venue":"Language Testing","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false},"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Psychology; Linguistics; Sociology; Philosophy","authors":[{"name":"Zhi Li","is_ca":true}],"retraction":null,"screen_n_in":null,"score":{"opus":0.01741045735498279,"gpt":0.3375408814479155,"spread":0.3201304240929327,"validation_status":"score_only:v0-immature-baseline"},"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00172693,0.001798997,0.003506903,0.005709479,0.0004855952,0.00342878,0.002267603,0.002424608,0.04336648],"category_scores_gemma":[0.008051196,0.0007211311,0.001272137,0.00860929,0.0007939931,0.003054019,0.001265589,0.003180069,0.0319678],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001829734,"about_ca_system_score_gemma":0.00431752,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005918715,"about_ca_topic_score_gemma":0.0135308,"domain_scores_codex":[0.9982389,0.0003202933,0.0002334703,0.0001918678,0.0009092348,0.0001061196],"domain_scores_gemma":[0.9929946,0.003281785,0.0007981328,0.000128954,0.002269938,0.0005265838],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00003516505,0.00002318775,0.0000764033,0.006506568,0.00004315454,0.00004767223,0.00004563451,0.0001138495,0.0001292279,0.0006548604,0.7447518,0.2475725],"study_design_scores_gemma":[0.00002003235,0.00003884306,0.0006147796,0.006078639,0.00005800838,0.000539795,0.00004555137,0.00005722386,0.000068397,0.0007622139,0.9916985,0.00001809965],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"review","genre_gemma":"other","genre_scores_codex":[0.00006622175,0.9838356,0.0002952583,0.003619943,0.007536019,0.00003198014,0.0001973147,0.00005194247,0.004365704],"genre_scores_gemma":[0.0006445792,0.9716892,0.0006175718,0.001909038,0.005554104,0.00009128982,0.0005186516,0.00003475193,0.01894085],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.04336648,"threshold_uncertainty_score":0.1450754,"prediction_status":"machine_predicted_unvalidated"},"labels":[],"label_agreement":null}]}