{"meta":{"query_hash":"9bc749b77fef","filters":{"venue":"Studies in Language Assessment"},"cohort_total":3,"direct_labels_cover":0,"predictions_cover":3,"exported":3,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/9bc749b77fef","api":"https://metacan.xera.ac/api/v1/cohort?venue=Studies+in+Language+Assessment"},"results":[{"id":"W4367592383","doi":"10.58379/homq5772","title":"Exploring shared and individual assessment of paired oral interactions","year":2022,"lang":"en","type":"article","venue":"Studies in Language Assessment","topic":"EFL/ESL Teaching and Learning","field":"Arts and Humanities","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Rubric; Psychology; Task (project management); Variation (astronomy); Modality (human–computer interaction); Audiology; Cognitive psychology; Computer science; Medicine; Artificial intelligence; Mathematics education","score_opus":0.3685810683728511,"score_gpt":0.42619052411325276,"score_spread":0.05760945574040166,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4367592383","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99068415,0.00013026778,0.0058115083,0.000025173684,0.0000125286915,0.00007178082,0.00007255848,0.00003587992,0.0031562306],"genre_scores_gemma":[0.9950265,0.00007098489,0.0040210825,0.000011434342,0.00000907299,0.000084256935,0.0000775342,0.00001538376,0.00068372715],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.98928356,0.005354604,0.0009180986,0.0012480753,0.0027357158,0.0004600575],"domain_scores_gemma":[0.96216196,0.021707872,0.006308435,0.0023960723,0.0059927464,0.0014329028],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00932718,0.00050562114,0.0006032001,0.0024148813,0.00060618436,0.0024785567,0.0007519381,0.00046764265,0.0021924358],"category_scores_gemma":[0.040921707,0.00024313269,0.0004022984,0.0007571854,0.001005651,0.0011872986,0.0038260734,0.0005798872,0.00041700518],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013907099,0.00055639696,0.5449762,0.0007459543,0.0007496392,0.0009478261,0.10277128,0.0015798352,0.05835819,0.00079850433,0.0007005083,0.28642496],"study_design_scores_gemma":[0.00003710448,0.0018979523,0.92352676,0.00014220212,0.000166378,0.0016530767,0.04202495,0.0052961498,0.02099677,0.0012423907,0.0028414968,0.0001747842],"about_ca_topic_score_codex":0.0012126553,"about_ca_topic_score_gemma":0.0030782423,"teacher_disagreement_score":0.00932718,"about_ca_system_score_codex":0.0004830698,"about_ca_system_score_gemma":0.00055472413,"threshold_uncertainty_score":0.049327374},"labels":[],"label_agreement":null},{"id":"W4376626656","doi":"10.58379/dayb9070","title":"Development of a Spanish generic writing skills scale for the Colombian Graduate Skills Assessment (Saber Pro)","year":2015,"lang":"en","type":"article","venue":"Studies in Language Assessment","topic":"Writing and Handwriting Education","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"La Trobe University","keywords":"Graduation (instrument); Scale (ratio); Context (archaeology); Mathematics education; Scripting language; Psychology; Trait; Test (biology); Writing assessment; Reliability (semiconductor); Pedagogy; Computer science; Geography; Mathematics; Cartography","score_opus":0.1354379703205154,"score_gpt":0.47668240693762326,"score_spread":0.34124443661710785,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4376626656","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.87678665,0.0013305964,0.04185243,0.0016644936,0.0005238595,0.011144576,0.0069607585,0.0009079588,0.05882862],"genre_scores_gemma":[0.76660997,0.001273581,0.20173022,0.00038598236,0.00008919252,0.012558937,0.005011203,0.00015179483,0.0121891275],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9983077,0.0005541006,0.00025561862,0.00014987771,0.00064267544,0.00008998938],"domain_scores_gemma":[0.9967429,0.0005892074,0.00042403213,0.00016435844,0.0018236239,0.00025586376],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032236378,0.0004946049,0.0003790556,0.0020154219,0.000412768,0.0008000348,0.0005660232,0.00029790675,0.0035174182],"category_scores_gemma":[0.009926713,0.00019043202,0.00036412638,0.000680071,0.00032779924,0.0004957989,0.00091719156,0.0006890584,0.0010842682],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020123208,0.00068311754,0.32665452,0.0010191076,0.00008646634,0.0006712145,0.0077973744,0.001333759,0.0124740945,0.0017745049,0.025552914,0.6217517],"study_design_scores_gemma":[0.00010434915,0.00057759153,0.9160292,0.00047455681,0.000045406574,0.0010440097,0.0060331007,0.004107222,0.0029848858,0.0011270111,0.067371234,0.00010134915],"about_ca_topic_score_codex":0.0069399066,"about_ca_topic_score_gemma":0.019995863,"teacher_disagreement_score":0.0069399066,"about_ca_system_score_codex":0.00090992084,"about_ca_system_score_gemma":0.002108906,"threshold_uncertainty_score":0.017048478},"labels":[],"label_agreement":null},{"id":"W4376626995","doi":"10.58379/rshg8366","title":"DIF investigations across groups of gender and academic background in a large-scale high-stakes language test ","year":2015,"lang":"en","type":"article","venue":"Studies in Language Assessment","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Queen's University","keywords":"Test (biology); Differential item functioning; Psychology; Quality (philosophy); Reliability (semiconductor); Scale (ratio); Gender bias; Social psychology; Applied psychology; Mathematics education; Medical education; Item response theory; Developmental psychology; Psychometrics; Medicine","score_opus":0.11975842518511098,"score_gpt":0.46502313377722004,"score_spread":0.34526470859210906,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4376626995","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9965592,0.000088328045,0.0020132111,0.00009053443,0.000024337005,0.00010035325,0.000057186815,0.0000060166717,0.0010607865],"genre_scores_gemma":[0.99842644,0.00002846117,0.0010995874,0.00005161251,0.000014133468,0.00008854158,0.00008057636,0.0000042118413,0.000206378],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.97159606,0.014780387,0.003445428,0.0019908238,0.0068048094,0.0013825218],"domain_scores_gemma":[0.91588247,0.045783307,0.011258354,0.007485604,0.017835164,0.0017551187],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.040404044,0.00045310234,0.00069807767,0.003450269,0.0012703352,0.0010714261,0.0007638202,0.00049195794,0.0014646762],"category_scores_gemma":[0.108169794,0.00019855378,0.0009474493,0.001651765,0.0016252495,0.0011933197,0.0021456613,0.0005769636,0.00026552432],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040372353,0.00026182513,0.94342405,0.00008044475,0.00015155991,0.00012015394,0.012631078,0.00013808614,0.0014411221,0.0007012446,0.00027356576,0.04037327],"study_design_scores_gemma":[0.00003085711,0.00073570077,0.98294604,0.00006945791,0.00007142502,0.00025581577,0.010265065,0.0013865526,0.0020003093,0.0010249966,0.0011773878,0.000036373403],"about_ca_topic_score_codex":0.0021361227,"about_ca_topic_score_gemma":0.00292848,"teacher_disagreement_score":0.040404044,"about_ca_system_score_codex":0.0010501702,"about_ca_system_score_gemma":0.001098234,"threshold_uncertainty_score":0.21367955},"labels":[],"label_agreement":null}]}