{"meta":{"query_hash":"ec1f2ca963e3","filters":{"venue":"Language Testing in Asia"},"cohort_total":12,"direct_labels_cover":0,"predictions_cover":12,"exported":12,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/ec1f2ca963e3","api":"https://metacan.xera.ac/api/v1/cohort?venue=Language+Testing+in+Asia"},"results":[{"id":"W1569132542","doi":"10.1186/s40468-015-0020-6","title":"Chinese university students’ perceptions of assessment tasks and classroom assessment environment","year":2015,"lang":"en","type":"article","venue":"Language Testing in Asia","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Psychology; Learning environment; Context (archaeology); Perception; Mathematics education; Formative assessment; Scale (ratio); Pedagogy","score_opus":0.028109092519047394,"score_gpt":0.3653477989662329,"score_spread":0.33723870644718545,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1569132542","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9992254,0.000031015512,0.00005200158,0.000042580254,0.000001964324,0.0000062824083,0.000015462221,0.0000017534907,0.0006237307],"genre_scores_gemma":[0.9994331,0.000056341603,0.000069894464,0.000025106781,0.0000020331574,0.000010398321,0.000031405092,8.021215e-7,0.00037086438],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9990977,0.00012350014,0.00011707957,0.00011187334,0.00034413583,0.00020583368],"domain_scores_gemma":[0.99745387,0.00049301283,0.0006858986,0.00012499053,0.0004686866,0.0007736095],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012087529,0.0003465667,0.0003127586,0.0008998459,0.0014709192,0.0015218019,0.00028987855,0.00039196853,0.0016907774],"category_scores_gemma":[0.0026465293,0.00016743116,0.0004249162,0.0013026189,0.00082095625,0.0006958864,0.0010455382,0.0004159662,0.00016486378],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006381631,0.00023657015,0.94005686,0.000086257045,0.000025026791,0.00024515958,0.029996242,0.00030554016,0.002576571,0.00044087984,0.0003867432,0.02558029],"study_design_scores_gemma":[0.0000046237187,0.00007906578,0.98559004,0.000018361749,0.000010637296,0.00005186734,0.012514851,0.00046541097,0.00035652696,0.00016490043,0.0007237588,0.000020090072],"about_ca_topic_score_codex":0.037828133,"about_ca_topic_score_gemma":0.050725877,"teacher_disagreement_score":0.037828133,"about_ca_system_score_codex":0.00151075,"about_ca_system_score_gemma":0.002411839,"threshold_uncertainty_score":0.075215936},"labels":[],"label_agreement":null},{"id":"W1921505225","doi":"10.1186/s40468-015-0018-0","title":"Raising the bar: language testing experience and second language motivation among South Korean young adolescents","year":2015,"lang":"en","type":"article","venue":"Language Testing in Asia","topic":"Multilingual Education and Policy","field":"Social Sciences","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; University of British Columbia","funders":"","keywords":"Psychology; Language assessment; Test (biology); Exploratory factor analysis; Salient; Context (archaeology); Language proficiency; Language education; Developmental psychology; Mathematics education; Pedagogy; Social psychology; Psychometrics","score_opus":0.10666736387976755,"score_gpt":0.4019436657537405,"score_spread":0.29527630187397297,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1921505225","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9998318,0.000025051202,0.00000821032,0.000019447794,9.41153e-7,0.0000018318249,0.0000044962158,2.3273788e-7,0.000107910826],"genre_scores_gemma":[0.9997738,0.00003774551,0.000022802684,0.000015584981,9.29565e-7,0.000004155902,0.000013911988,4.000244e-7,0.00013061186],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9996637,0.00008170991,0.000036764835,0.00004080844,0.00008778827,0.00008908064],"domain_scores_gemma":[0.9982256,0.0003177807,0.0007816723,0.00003620408,0.0001634097,0.00047532603],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009889511,0.00020154787,0.00020862096,0.00064147695,0.0006079412,0.0010698166,0.00025493154,0.00043943033,0.0014320244],"category_scores_gemma":[0.0019799797,0.00024512585,0.0003983792,0.0003553859,0.00049877685,0.0006858259,0.00082898827,0.0007294804,0.000127611],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000018204835,0.00013982107,0.9877588,0.000021354424,0.000010557404,0.00018463266,0.008283738,0.000011720949,0.0004965595,0.00006982342,0.00005648894,0.002948364],"study_design_scores_gemma":[0.0000021452227,0.00010054592,0.97410834,0.000026136282,0.0000144117075,0.0001867077,0.024864323,0.0000931283,0.00019670132,0.000033854067,0.0003661351,0.000007614423],"about_ca_topic_score_codex":0.0040902854,"about_ca_topic_score_gemma":0.009049878,"teacher_disagreement_score":0.0040902854,"about_ca_system_score_codex":0.00039101872,"about_ca_system_score_gemma":0.0006698396,"threshold_uncertainty_score":0.008132935},"labels":[],"label_agreement":null},{"id":"W2112346409","doi":"10.1186/s40468-015-0016-2","title":"How language proficiency contributes to Chinese students’ academic success in Korean universities","year":2015,"lang":"en","type":"article","venue":"Language Testing in Asia","topic":"Grit, Self-Efficacy, and Motivation","field":"Psychology","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Psychology; Language proficiency; Test of English as a Foreign Language; Academic achievement; Medical education; Graduate students; Mathematics education; English language; Pedagogy; Medicine","score_opus":0.03571550564783218,"score_gpt":0.36311434997557873,"score_spread":0.3273988443277466,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2112346409","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9995957,0.000019712865,0.000016489152,0.000024529196,0.0000013298743,0.0000019039272,0.000012132299,6.3938603e-7,0.00032762598],"genre_scores_gemma":[0.9998023,0.000028595514,0.00001261594,0.000005910937,0.0000013858245,0.0000015396118,0.000017727629,4.9236695e-7,0.00012933144],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9996382,0.00006409139,0.000049697166,0.00005571926,0.00007399999,0.000118308395],"domain_scores_gemma":[0.99775285,0.0003766158,0.0006413667,0.0001013407,0.00021627556,0.0009116298],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007857824,0.0003363858,0.00025707312,0.00087274093,0.00062813715,0.0011134762,0.0002538963,0.00026090533,0.0022581103],"category_scores_gemma":[0.002584395,0.00016648743,0.00048626348,0.00078985956,0.0004858192,0.00053505535,0.00085526844,0.0004435993,0.00027849045],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000028077304,0.0000762119,0.9947466,0.000011751751,0.000032031938,0.00013946775,0.0008268394,0.00008157425,0.0004773021,0.00006728288,0.000056703866,0.0034560764],"study_design_scores_gemma":[0.000001551817,0.00003893178,0.9980567,0.0000064088467,0.0000132083815,0.000035505203,0.0013770546,0.0002399173,0.00009975274,0.000046587153,0.00007938206,0.0000050628532],"about_ca_topic_score_codex":0.010528441,"about_ca_topic_score_gemma":0.016040921,"teacher_disagreement_score":0.010528441,"about_ca_system_score_codex":0.000373999,"about_ca_system_score_gemma":0.0009644313,"threshold_uncertainty_score":0.020934343},"labels":[],"label_agreement":null},{"id":"W2886638441","doi":"10.1186/s40468-018-0065-4","title":"How does anxiety influence language performance? From the perspectives of foreign language classroom anxiety and cognitive test anxiety","year":2018,"lang":"en","type":"article","venue":"Language Testing in Asia","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":175,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Anxiety; Psychology; Test anxiety; Foreign language; Test (biology); Foreign language anxiety; Cognition; Developmental psychology; Language assessment; Clinical psychology; Consistency (knowledge bases); Mathematics education; Psychiatry","score_opus":0.012529891190225325,"score_gpt":0.2365920863830132,"score_spread":0.2240621951927879,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2886638441","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9935528,0.0009810339,0.00033552267,0.0012256936,0.000022037051,0.000009093848,0.000033666514,0.000007058117,0.0038330536],"genre_scores_gemma":[0.99947125,0.00024418795,0.000044705932,0.00007014133,0.000022503318,0.000002642954,0.000010878916,0.0000011053611,0.00013243845],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9983802,0.0007746943,0.0000691162,0.00011420139,0.00034589163,0.00031591766],"domain_scores_gemma":[0.99240935,0.004269899,0.0016231611,0.00013177373,0.00042182373,0.0011439708],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012652086,0.00027405043,0.0002999619,0.0007702711,0.00041107027,0.0021615082,0.00026812777,0.00050683686,0.0011350979],"category_scores_gemma":[0.0070453547,0.00014942663,0.00053555815,0.0004173438,0.0012349291,0.0007531242,0.0006135324,0.00094663276,0.000112255184],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015495691,0.00033339567,0.96746874,0.00006211428,0.00014957662,0.0008163624,0.006134834,0.00014125062,0.0012159651,0.0008742405,0.00024089927,0.022407599],"study_design_scores_gemma":[0.0000100068555,0.00031211184,0.9907706,0.00003861356,0.00010546203,0.000503527,0.0062085823,0.00045904118,0.00027325333,0.00068182795,0.0006137532,0.0000232873],"about_ca_topic_score_codex":0.0032757607,"about_ca_topic_score_gemma":0.0035571628,"teacher_disagreement_score":0.0032757607,"about_ca_system_score_codex":0.0005755756,"about_ca_system_score_gemma":0.00086748326,"threshold_uncertainty_score":0.006691158},"labels":[],"label_agreement":null},{"id":"W2923578044","doi":"10.1186/s40468-019-0078-7","title":"The language assessment literacy needs of Iranian EFL teachers with a focus on reformed assessment policies","year":2019,"lang":"en","type":"article","venue":"Language Testing in Asia","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":59,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Rubric; Psychology; Literacy; Curriculum; Pedagogy; Active listening; Mathematics education; Language assessment; Alternative assessment; Perception; Focus group; Sociology","score_opus":0.016486939779852855,"score_gpt":0.3597537479867195,"score_spread":0.3432668082068666,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2923578044","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9967217,0.00015775772,0.00006030128,0.0015661052,0.000006430713,0.000009949355,0.000014030252,0.0000034721643,0.0014602217],"genre_scores_gemma":[0.9990589,0.00012593348,0.00013245428,0.00021348016,0.000004504955,0.0000120731665,0.000015322961,0.0000012037613,0.00043616307],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.997486,0.00067053316,0.00020398683,0.00014069487,0.00069567596,0.00080321066],"domain_scores_gemma":[0.99199706,0.0020704805,0.0020372777,0.00017931557,0.0027901523,0.0009256622],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003266443,0.00013949341,0.00027468058,0.0007862643,0.00136874,0.0017656697,0.0004048916,0.0008837045,0.0011923789],"category_scores_gemma":[0.013325224,0.00019501218,0.00015295127,0.00072168483,0.0010380455,0.0018206025,0.0009734715,0.0009819649,0.0002046536],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019710383,0.0006450421,0.5118874,0.0003284755,0.000013316358,0.0022253285,0.37258172,0.0004047985,0.0037561376,0.0015951028,0.002323462,0.10404214],"study_design_scores_gemma":[0.00002139716,0.00041223245,0.41187942,0.00021470779,0.000012426173,0.0012072541,0.5721058,0.0007444667,0.0009969135,0.00080845854,0.011544034,0.000052965603],"about_ca_topic_score_codex":0.018954339,"about_ca_topic_score_gemma":0.023349416,"teacher_disagreement_score":0.018954339,"about_ca_system_score_codex":0.0028143574,"about_ca_system_score_gemma":0.0056414404,"threshold_uncertainty_score":0.037688017},"labels":[],"label_agreement":null},{"id":"W2967888896","doi":"10.1186/s40468-019-0089-4","title":"Critical review of validation models and practices in language testing: their limitations and future directions for validation research","year":2019,"lang":"en","type":"article","venue":"Language Testing in Asia","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":46,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Argument (complex analysis); Empirical research; Construct (python library); Computer science; Test (biology); Construct validity; Language assessment; Management science; Psychology; Data science; Psychometrics; Statistics; Mathematics education; Mathematics; Engineering","score_opus":0.2871225134574545,"score_gpt":0.4973541369068704,"score_spread":0.21023162344941593,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2967888896","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0048391228,0.8135438,0.043423664,0.114996694,0.010075686,0.0022562377,0.0004166136,0.0002141802,0.010233916],"genre_scores_gemma":[0.1050491,0.73127675,0.10563485,0.04156501,0.005088027,0.008264305,0.00066071353,0.0004891545,0.0019721275],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.5562592,0.31613237,0.06128142,0.008355281,0.055803195,0.00216854],"domain_scores_gemma":[0.11774201,0.70557404,0.027631246,0.023806734,0.123523556,0.0017224522],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.447146,0.0019351265,0.0044118306,0.03158281,0.005530765,0.015263579,0.006665227,0.00573032,0.0032001443],"category_scores_gemma":[0.7135997,0.001975636,0.0038540876,0.023809375,0.016306981,0.023742698,0.007797544,0.010353054,0.0013213954],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026552155,0.00009912292,0.0032671406,0.1231972,0.0009448603,0.00045379682,0.03095689,0.0008516443,0.00060319743,0.09072774,0.0700017,0.67863125],"study_design_scores_gemma":[0.00012051671,0.00021786032,0.0040289257,0.47346228,0.0013057304,0.0006501459,0.019912424,0.001425595,0.0016883989,0.054177996,0.44277218,0.00023788032],"about_ca_topic_score_codex":0.008780839,"about_ca_topic_score_gemma":0.00958327,"teacher_disagreement_score":0.552854,"about_ca_system_score_codex":0.022100002,"about_ca_system_score_gemma":0.06731454,"threshold_uncertainty_score":0.68176746},"labels":[],"label_agreement":null},{"id":"W2984367764","doi":"10.1186/s40468-019-0094-7","title":"Assessing peer review pattern and the effect of face-to-face and mobile-mediated modes on students’ academic writing development","year":2019,"lang":"en","type":"article","venue":"Language Testing in Asia","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Psychology; Academic writing; English for academic purposes; Peer feedback; Mathematics education; Cohesion (chemistry); Task (project management); Second language writing; Face-to-face; Pedagogy; Medical education; Second language; Linguistics","score_opus":0.032727561803754454,"score_gpt":0.399730670787874,"score_spread":0.36700310898411953,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2984367764","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9960078,0.00014901425,0.001869552,0.000118426804,0.000025628817,0.00017819708,0.000024486226,0.00007220599,0.0015546577],"genre_scores_gemma":[0.9953999,0.00008798263,0.0032345532,0.000039425762,0.000029895871,0.00021197922,0.00002737712,0.000018637234,0.0009502279],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.97239673,0.015456316,0.0026280321,0.0023121613,0.0066151563,0.0005916316],"domain_scores_gemma":[0.80783755,0.12968504,0.025277263,0.011345122,0.020386286,0.0054686978],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.015401662,0.00040196278,0.0008098075,0.0015209927,0.0007787039,0.0017541727,0.001127828,0.0005401295,0.0024847265],"category_scores_gemma":[0.15419523,0.0003104045,0.00035819225,0.00062355085,0.00061114546,0.0013878292,0.0017822753,0.00059482653,0.00083849626],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028453756,0.0019073712,0.45699933,0.00064887793,0.00030094985,0.0005502063,0.019994166,0.00074927096,0.02742742,0.0002049456,0.0012510278,0.48712105],"study_design_scores_gemma":[0.0002608212,0.0060189096,0.9556075,0.00018850512,0.00022545666,0.0010590535,0.01633329,0.0045891157,0.011604193,0.000458079,0.0035199313,0.00013510916],"about_ca_topic_score_codex":0.0008066833,"about_ca_topic_score_gemma":0.0016507454,"teacher_disagreement_score":0.98459834,"about_ca_system_score_codex":0.00053458684,"about_ca_system_score_gemma":0.0007746325,"threshold_uncertainty_score":0.08145273},"labels":[],"label_agreement":null},{"id":"W4309467076","doi":"10.1186/s40468-022-00201-5","title":"Lessons from the Chinese imperial examination system","year":2022,"lang":"en","type":"article","venue":"Language Testing in Asia","topic":"Global Educational Reforms and Inequalities","field":"Social Sciences","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Imperial examination; Language assessment; Linguistics; Set (abstract data type); Psychology; Field (mathematics); History; Pedagogy; Computer science; Philosophy; Ancient history","score_opus":0.031570842347215446,"score_gpt":0.34437883217476145,"score_spread":0.312807989827546,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4309467076","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.70500654,0.0015935311,0.0038203134,0.08625294,0.0005138896,0.00014640414,0.00025234275,0.00010568573,0.20230822],"genre_scores_gemma":[0.992585,0.0001672906,0.00064501504,0.0013111451,0.000063152605,0.00003952683,0.000047747253,0.000011409353,0.0051295944],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99315506,0.0029385136,0.00030893594,0.00060612743,0.0013628526,0.0016286131],"domain_scores_gemma":[0.99397284,0.0020358276,0.0004065164,0.00096334727,0.0016190158,0.0010023271],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008311572,0.00029849497,0.00040484546,0.0018606741,0.005405618,0.0040786937,0.0015275177,0.0012001067,0.004486378],"category_scores_gemma":[0.015201535,0.00019659629,0.00033847764,0.0021107902,0.013416971,0.0025271322,0.0054499726,0.002942919,0.00032972213],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000118967924,0.00007652163,0.047159407,0.00014275905,0.000028713213,0.0012744329,0.055444248,0.00085787545,0.0003743565,0.8128351,0.015276264,0.06641138],"study_design_scores_gemma":[0.00024024435,0.00035122962,0.27608955,0.000989992,0.00015530968,0.0013072732,0.07214592,0.010185205,0.0026113119,0.2996618,0.33599102,0.00027118876],"about_ca_topic_score_codex":0.19185375,"about_ca_topic_score_gemma":0.10895495,"teacher_disagreement_score":0.19185375,"about_ca_system_score_codex":0.017410642,"about_ca_system_score_gemma":0.01900195,"threshold_uncertainty_score":0.38147408},"labels":[],"label_agreement":null},{"id":"W4387821100","doi":"10.1186/s40468-023-00261-1","title":"Instructional practices and students’ reading performance: a comparative study of 10 top performing regions in PISA 2018","year":2023,"lang":"en","type":"article","venue":"Language Testing in Asia","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Enthusiasm; Psychology; Reading (process); Multilevel model; Mathematics education; China; Sample (material); Teacher education; Pedagogy; Geography; Social psychology; Political science; Chemistry","score_opus":0.1189039487434486,"score_gpt":0.4392416959406387,"score_spread":0.3203377471971901,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387821100","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9997054,0.000022014918,0.00001845852,0.0000073644665,6.6826607e-7,0.000003560123,0.000039490285,0.0000012476868,0.00020181373],"genre_scores_gemma":[0.999686,0.00003887251,0.000063112384,0.000008855894,0.0000010842941,0.00001164355,0.000087754495,9.706904e-7,0.000101830294],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9995177,0.00011919225,0.00004263176,0.00010309112,0.00007663698,0.00014071265],"domain_scores_gemma":[0.9989015,0.00025814303,0.00031877228,0.00008730638,0.00019756398,0.00023676948],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006247242,0.00027354844,0.0005287293,0.0018096081,0.0012148421,0.0009505208,0.0003931451,0.0002682922,0.0011128309],"category_scores_gemma":[0.001547433,0.00031068633,0.000357505,0.002376572,0.0005529578,0.00046811387,0.0010720136,0.0004331927,0.0002831244],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007699221,0.00019045689,0.9836248,0.000026976257,0.000038208676,0.0002314762,0.007899558,0.000043916094,0.00082489307,0.000026206822,0.00008379132,0.0069328253],"study_design_scores_gemma":[0.0000017460148,0.00010694511,0.9918276,0.0000066861735,0.000013851708,0.00006716956,0.0076119085,0.000039615723,0.00016666137,0.0000050018793,0.00014967051,0.0000032369276],"about_ca_topic_score_codex":0.02373951,"about_ca_topic_score_gemma":0.04453219,"teacher_disagreement_score":0.02373951,"about_ca_system_score_codex":0.00086365803,"about_ca_system_score_gemma":0.00094765064,"threshold_uncertainty_score":0.047202647},"labels":[],"label_agreement":null},{"id":"W4400514459","doi":"10.1186/s40468-024-00294-0","title":"Adaptation and norm determination of the Boston Naming Test for healthy Lebanese adults aged between 50 and 88 years","year":2024,"lang":"en","type":"article","venue":"Language Testing in Asia","topic":"Neurobiology of Language and Bilingualism","field":"Neuroscience","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Normative; Boston Naming Test; Psychology; Norm (philosophy); Test (biology); Cognition; Clinical psychology; Developmental psychology; Gerontology; Neuropsychology; Medicine; Psychiatry","score_opus":0.040891310306870876,"score_gpt":0.30745301490376475,"score_spread":0.2665617045968939,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400514459","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9959026,0.000091975126,0.0019085145,0.00003062192,0.000026228061,0.00023407179,0.00038656394,0.000018069364,0.0014014345],"genre_scores_gemma":[0.992595,0.00009599744,0.0046926024,0.000054538672,0.000019057692,0.0004846574,0.0012520482,0.000014932756,0.00079127465],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9990538,0.0003080161,0.00021589128,0.00015043208,0.00022339707,0.00004837497],"domain_scores_gemma":[0.997672,0.0004484431,0.0003905134,0.00027030785,0.0011053984,0.00011327842],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002414035,0.0003587373,0.00025974849,0.0009956036,0.0004347971,0.00037141392,0.00030690362,0.00030767414,0.0012059325],"category_scores_gemma":[0.0063170013,0.00011479711,0.00027355627,0.00038249913,0.00025658996,0.00036773115,0.00044719907,0.0002683321,0.00051027274],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005553114,0.0004995384,0.94115514,0.000067967514,0.00004798352,0.00046884659,0.0019863662,0.00027578475,0.009605944,0.00026101817,0.0010533888,0.044022676],"study_design_scores_gemma":[0.000031264677,0.0006382701,0.9942058,0.000018647515,0.000016883441,0.0006031442,0.0008404866,0.0005972884,0.0014091096,0.00010681874,0.0015198954,0.000012383742],"about_ca_topic_score_codex":0.0029088033,"about_ca_topic_score_gemma":0.004858697,"teacher_disagreement_score":0.0029088033,"about_ca_system_score_codex":0.00034026225,"about_ca_system_score_gemma":0.00039072582,"threshold_uncertainty_score":0.0127667785},"labels":[],"label_agreement":null},{"id":"W4407613621","doi":"10.1186/s40468-025-00341-4","title":"Correction: Lessons from the Chinese imperial examination system","year":2025,"lang":"en","type":"article","venue":"Language Testing in Asia","topic":"Legal Education and Practice Innovations","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Imperial examination; Linguistics; Psychology; Language education; Mathematics education; History; Philosophy; Ancient history","score_opus":0.027878936346111008,"score_gpt":0.3862787152161436,"score_spread":0.3583997788700326,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407613621","genre_codex":"commentary","genre_gemma":"editorial","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"editorial","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017176581,0.0013786615,0.0013794223,0.507303,0.46701515,0.000049569302,0.0011436468,0.0003881417,0.019624777],"genre_scores_gemma":[0.1240414,0.0051494245,0.006060999,0.30288768,0.24757698,0.00030936068,0.0009775601,0.00095593306,0.3120407],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99411124,0.001275682,0.0010788061,0.0007751723,0.002063374,0.00069581624],"domain_scores_gemma":[0.9420259,0.017374506,0.0023394397,0.0043715322,0.031260174,0.0026285737],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0066004153,0.00096369284,0.0009668906,0.0022859091,0.004257204,0.003973119,0.003448761,0.010052881,0.025105095],"category_scores_gemma":[0.14082298,0.000568132,0.0006096326,0.002327006,0.0049274554,0.0029162546,0.0021905932,0.014331721,0.010013816],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003069373,0.0000033089718,0.00042324155,0.00006778509,0.000011653981,0.0006046528,0.0005055478,0.000047446527,0.000017972932,0.009326881,0.9818942,0.0070666675],"study_design_scores_gemma":[0.000056174005,0.000012667238,0.0026157468,0.00045606447,0.000034994144,0.0010240247,0.0007930656,0.00042065346,0.00024816688,0.006787663,0.98748624,0.00006443625],"about_ca_topic_score_codex":0.09448743,"about_ca_topic_score_gemma":0.08764252,"teacher_disagreement_score":0.09448743,"about_ca_system_score_codex":0.0069408864,"about_ca_system_score_gemma":0.011634243,"threshold_uncertainty_score":0.18787491},"labels":[],"label_agreement":null},{"id":"W4415055623","doi":"10.1186/s40468-025-00390-9","title":"Shaping written corrective feedback perspectives and practices: comparing novice and experienced instructors of English for academic purposes in Bangladesh","year":2025,"lang":"en","type":"article","venue":"Language Testing in Asia","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba; Red River College","funders":"","keywords":"Corrective feedback; Professional development; English for academic purposes; Grammar; Faculty development; Relevance (law); Focus group; Qualitative research; Higher education","score_opus":0.06050911231811046,"score_gpt":0.33303835002998666,"score_spread":0.2725292377118762,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415055623","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9962529,0.00017561852,0.0005196016,0.00045483437,0.000011861358,0.000034041437,0.000027347796,0.000008668939,0.002515196],"genre_scores_gemma":[0.9968098,0.00023890792,0.00035001538,0.00018688737,0.0000041620924,0.000036872425,0.00001958814,0.000012500498,0.0023412388],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9958895,0.0024304085,0.00024060001,0.00033905197,0.00061859033,0.00048179814],"domain_scores_gemma":[0.9874117,0.008063946,0.001313213,0.00040903012,0.0013147932,0.0014873886],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039816233,0.00034390012,0.0004146591,0.0008081602,0.0030178372,0.0024980095,0.000835458,0.0009524608,0.0030769024],"category_scores_gemma":[0.017693102,0.00034782794,0.00015000223,0.00075040443,0.0036627376,0.0015920965,0.0022782062,0.0012669705,0.0004858495],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000043622334,0.000055856108,0.014768284,0.00012561884,0.0000028619565,0.0010740493,0.9705695,0.000026387224,0.0033149086,0.00028125578,0.00027013457,0.009467562],"study_design_scores_gemma":[0.000005959381,0.00010708993,0.008674823,0.00011246971,0.0000034796647,0.00041683568,0.98414385,0.000056412515,0.00061953726,0.000103312006,0.005736601,0.000019601743],"about_ca_topic_score_codex":0.0077245617,"about_ca_topic_score_gemma":0.014446548,"teacher_disagreement_score":0.0077245617,"about_ca_system_score_codex":0.0026959595,"about_ca_system_score_gemma":0.0027266133,"threshold_uncertainty_score":0.02105707},"labels":[],"label_agreement":null}]}