{"meta":{"query_hash":"3be08ccc75e3","filters":{"venue":"Assessment in Education Principles Policy and Practice"},"cohort_total":30,"direct_labels_cover":0,"predictions_cover":30,"exported":30,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/3be08ccc75e3","api":"https://metacan.xera.ac/api/v1/cohort?venue=Assessment+in+Education+Principles+Policy+and+Practice"},"results":[{"id":"W1978227392","doi":"10.1080/09695940903565362","title":"Teacher beliefs about the cognitive diagnostic information of classroom‐ versus large‐scale tests: implications for assessment literacy","year":2010,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":58,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Smiths Detection (Canada); University of Alberta","funders":"","keywords":"Test (biology); Mathematics education; Scale (ratio); Psychology; Literacy; Pedagogy","score_opus":0.05055241862711323,"score_gpt":0.48032179247732004,"score_spread":0.4297693738502068,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1978227392","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99188614,0.00024301311,0.0011651815,0.0019535585,0.000009878112,0.000032793072,0.00001531575,0.000005557,0.0046885805],"genre_scores_gemma":[0.9993843,0.00008268609,0.00025888154,0.000109025896,0.000004612682,0.000013857664,0.0000060101356,0.0000014053752,0.00013923016],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98628485,0.009205063,0.0006611623,0.00053826824,0.0027811956,0.0005295246],"domain_scores_gemma":[0.84626615,0.12573156,0.014900184,0.0026642012,0.0075896927,0.0028482473],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018061748,0.00013510532,0.00031527245,0.00089983945,0.0007124273,0.0021008593,0.00038190634,0.0005893644,0.0016410751],"category_scores_gemma":[0.099560484,0.00018641462,0.00023702162,0.0005637105,0.0026847508,0.0016724197,0.0011950225,0.0012352631,0.00020267752],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003199617,0.000556424,0.8799813,0.00017500063,0.000044236538,0.00020928221,0.051796883,0.00017874126,0.0012654676,0.002176912,0.00043723694,0.06285856],"study_design_scores_gemma":[0.00005866946,0.00097214413,0.91199565,0.00046277975,0.000072134186,0.0003476619,0.07655937,0.0013056658,0.0018964044,0.0032486925,0.0030447752,0.000036115744],"about_ca_topic_score_codex":0.004040514,"about_ca_topic_score_gemma":0.0056788507,"teacher_disagreement_score":0.018061748,"about_ca_system_score_codex":0.0014630883,"about_ca_system_score_gemma":0.0020025526,"threshold_uncertainty_score":0.095520735},"labels":[],"label_agreement":null},{"id":"W1996435010","doi":"10.1080/09695940701272773","title":"Did we take the same test? Differing accounts of the Ontario Secondary School Literacy Test by first and second language test‐takers","year":2007,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":74,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Queen's University; Carleton University","funders":"","keywords":"Test (biology); Graduation (instrument); Psychology; Context (archaeology); Construct (python library); Literacy; Construct validity; Mathematics education; Fidelity; Test score; Standardized test; Pedagogy; Developmental psychology; Computer science; Psychometrics; Mathematics","score_opus":0.02100229501179931,"score_gpt":0.38458132868967815,"score_spread":0.36357903367787886,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1996435010","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92587674,0.0020341796,0.0040572467,0.018252725,0.00013749169,0.00015667506,0.00035459912,0.000055695153,0.04907459],"genre_scores_gemma":[0.9910364,0.0006404976,0.0012047127,0.0014264357,0.000035411642,0.000058786205,0.00014571958,0.00005185678,0.0054000383],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9688408,0.011459429,0.0014289977,0.0017972499,0.014397805,0.0020757301],"domain_scores_gemma":[0.9702678,0.011937295,0.005158749,0.0027809558,0.0078006326,0.002054555],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015241263,0.00048024894,0.0004578446,0.0030776924,0.006924826,0.006276983,0.002250515,0.0016822041,0.0013975379],"category_scores_gemma":[0.062726915,0.00045078286,0.0006178707,0.002554065,0.016923836,0.0039197095,0.004454658,0.0024452854,0.00029282848],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008083533,0.00004395532,0.09512686,0.000111550624,0.000033325407,0.00063555536,0.86300915,0.00010999236,0.0010131459,0.015320596,0.0025560982,0.02195892],"study_design_scores_gemma":[0.000029515004,0.0001821476,0.3444324,0.0006555044,0.00007416746,0.0010197924,0.5689775,0.0011145981,0.0015667365,0.009352142,0.07239216,0.0002033274],"about_ca_topic_score_codex":0.72565645,"about_ca_topic_score_gemma":0.7678544,"teacher_disagreement_score":0.72565645,"about_ca_system_score_codex":0.03391194,"about_ca_system_score_gemma":0.01786655,"threshold_uncertainty_score":0.5519184},"labels":[],"label_agreement":null},{"id":"W2013464569","doi":"10.1080/0969594x.2013.776943","title":"Fair and equitable assessment practices for all students","year":2013,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":71,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Thompson Rivers University; University of Lethbridge; University of Alberta; University of Calgary","funders":"","keywords":"Intrusiveness; Equity (law); Psychology; Public relations; Best practice; Medical education; Applied psychology; Pedagogy; Social psychology; Political science; Medicine","score_opus":0.10231687439942877,"score_gpt":0.5279200270248844,"score_spread":0.4256031526254556,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2013464569","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.65144926,0.001850122,0.11632229,0.07695903,0.00066359184,0.002417034,0.0001831516,0.001277564,0.14887792],"genre_scores_gemma":[0.9596473,0.00028479268,0.03225647,0.001603969,0.00006321611,0.0003984181,0.000043947326,0.00004973496,0.005652059],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.86074704,0.09396933,0.00852835,0.0043170634,0.02781846,0.0046198033],"domain_scores_gemma":[0.84342015,0.048043232,0.016955202,0.031935066,0.04756581,0.012080552],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06826248,0.0004899547,0.000748409,0.0030492682,0.007909843,0.008764635,0.0029354065,0.0029372398,0.0034946383],"category_scores_gemma":[0.18625376,0.0003797446,0.00048840616,0.0021531917,0.003962099,0.007811042,0.012702419,0.0039201877,0.001094227],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015906732,0.0012634157,0.07330026,0.00034779502,0.00004293047,0.00028758988,0.08340397,0.0021184923,0.0028594097,0.045729678,0.015377735,0.7751097],"study_design_scores_gemma":[0.0001616588,0.0022520528,0.17419362,0.004081983,0.0000940471,0.0020195965,0.18448268,0.010308879,0.015773902,0.2698782,0.33609122,0.00066211587],"about_ca_topic_score_codex":0.0038677263,"about_ca_topic_score_gemma":0.008385749,"teacher_disagreement_score":0.06826248,"about_ca_system_score_codex":0.0049416954,"about_ca_system_score_gemma":0.018133085,"threshold_uncertainty_score":0.3610108},"labels":[],"label_agreement":null},{"id":"W2013643140","doi":"10.1080/0969594x.2014.953910","title":"The effects of key demographic variables on markers’ perceived ease of use and acceptance of onscreen marking","year":2014,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Hearing Impairment and Communication","field":"Psychology","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"AGE-WELL","keywords":"Rasch model; Psychology; Usability; Developmental psychology; Computer science","score_opus":0.03750763857916641,"score_gpt":0.39032217685889037,"score_spread":0.352814538279724,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2013643140","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9993299,0.000047055182,0.00011655126,0.00003179632,0.0000030257254,0.0000060470693,0.000022565206,0.000001968956,0.0004411889],"genre_scores_gemma":[0.99953663,0.000042986736,0.00013272099,0.000012928654,0.0000033904582,0.000009569818,0.000030768326,0.0000012635131,0.00022985162],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9981529,0.000857925,0.0001904996,0.0001467521,0.00047153825,0.00018036707],"domain_scores_gemma":[0.97809243,0.009787025,0.008140841,0.0007401251,0.001369846,0.0018697167],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023160416,0.00021447263,0.00019448221,0.0005876138,0.00024542492,0.00068901904,0.00019309364,0.00026978867,0.0025402266],"category_scores_gemma":[0.015993057,0.00014780492,0.0002582183,0.00034070874,0.00042860868,0.0005916556,0.00062784087,0.0004693743,0.00037777817],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022846968,0.0002531081,0.9828516,0.000040270148,0.00003495419,0.00016932195,0.004892214,0.000053598036,0.0011911828,0.000040446306,0.000070178554,0.010174588],"study_design_scores_gemma":[0.0000020617388,0.000300609,0.99520904,0.000011430704,0.000009791083,0.00014407554,0.003675011,0.00009705874,0.00026105586,0.000026418309,0.0002543399,0.000009047297],"about_ca_topic_score_codex":0.00084173476,"about_ca_topic_score_gemma":0.0017351247,"teacher_disagreement_score":0.0025402266,"about_ca_system_score_codex":0.00018661076,"about_ca_system_score_gemma":0.00026946992,"threshold_uncertainty_score":0.012248576},"labels":[],"label_agreement":null},{"id":"W2026041656","doi":"10.1080/09695940802164226","title":"Educational assessment in Canada","year":2008,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Educational Assessment and Improvement","field":"Decision Sciences","cited_by":66,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Brock University","funders":"Social Sciences and Humanities Research Council of Canada; Brock University","keywords":"Accountability; Independence (probability theory); Bureaucracy; Political science; Scale (ratio); Public administration; Multiculturalism; Educational assessment; Regional science; Economic growth; Geography; Sociology; Pedagogy; Economics; Law","score_opus":0.1406251533691962,"score_gpt":0.4873947963953773,"score_spread":0.3467696430261811,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2026041656","genre_codex":"other","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03965514,0.0075025437,0.0016287976,0.046310518,0.0020115816,0.00028206926,0.0034847036,0.0004218002,0.8987029],"genre_scores_gemma":[0.37565458,0.007497664,0.0045221443,0.010611626,0.00029704403,0.00018387593,0.002188834,0.00018981122,0.5988544],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.99693006,0.0002503415,0.00009274568,0.0002546971,0.0012924598,0.0011797844],"domain_scores_gemma":[0.9913545,0.0004608839,0.00016801205,0.00017874611,0.00438353,0.0034542696],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017509285,0.00038708624,0.0003408819,0.0035257179,0.015679287,0.0076081804,0.0015402018,0.0013028509,0.036598854],"category_scores_gemma":[0.006228313,0.00029209824,0.00045820215,0.005445746,0.0028612074,0.0016054446,0.0039314837,0.002386368,0.0041597188],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":true,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000065433116,0.000118124146,0.019104904,0.00030639363,0.000013744227,0.0007780292,0.007917128,0.000720863,0.0002974099,0.15499857,0.46978486,0.3458946],"study_design_scores_gemma":[0.000007747958,0.000011447356,0.021478675,0.00018694159,0.0000057807265,0.00010760941,0.0031192505,0.0002640866,0.00008318457,0.0022069912,0.9724966,0.000031633455],"about_ca_topic_score_codex":0.9915814,"about_ca_topic_score_gemma":0.99550325,"teacher_disagreement_score":0.8715099,"about_ca_system_score_codex":0.12849006,"about_ca_system_score_gemma":0.3434774,"threshold_uncertainty_score":0.932265},"labels":[],"label_agreement":null},{"id":"W2068327470","doi":"10.1080/09695940802164218","title":"Self‐assessment in a technology‐supported environment: the case of grade 9 geography","year":2008,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Gender and Technology in Education","field":"Social Sciences","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"University of Toronto; Institute for Christian Studies","funders":"University of Toronto","keywords":"Self-efficacy; Psychology; Variance (accounting); Self-assessment; Perception; Outcome (game theory); Treatment and control groups; Applied psychology; Medical education; Mathematics education; Social psychology; Medicine; Statistics; Mathematics; Accounting","score_opus":0.040487345031156426,"score_gpt":0.4035206507773588,"score_spread":0.3630333057462024,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2068327470","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9313449,0.00016501987,0.0015415327,0.0063311094,0.000041796517,0.00004581348,0.00002881921,0.000017110939,0.060484],"genre_scores_gemma":[0.99522865,0.00007558558,0.0005323365,0.000103658065,0.0000041319695,0.000013765104,0.00000900155,0.0000054754532,0.004027361],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.99730384,0.001815688,0.00007298684,0.00014156548,0.00019818323,0.00046772047],"domain_scores_gemma":[0.9955446,0.002268601,0.00034327467,0.00030986711,0.00056062615,0.000973049],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042028306,0.00020958873,0.00028654028,0.00071263715,0.0047550374,0.0056754006,0.0010779912,0.0019424557,0.0057706498],"category_scores_gemma":[0.010400597,0.00015775092,0.0003675417,0.0006732611,0.0040753097,0.0024283158,0.0042751306,0.0016798074,0.0006065444],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015813978,0.0010182441,0.15572402,0.0001186635,0.00004680531,0.03377104,0.50296307,0.002509203,0.0007075349,0.20460778,0.009737002,0.08863849],"study_design_scores_gemma":[0.00006290393,0.00035804338,0.0848043,0.000405237,0.000051875773,0.0076041752,0.7192836,0.00703001,0.0008267655,0.056078628,0.12339909,0.00009538187],"about_ca_topic_score_codex":0.05719899,"about_ca_topic_score_gemma":0.08012958,"teacher_disagreement_score":0.05719899,"about_ca_system_score_codex":0.004231481,"about_ca_system_score_gemma":0.0039423667,"threshold_uncertainty_score":0.11373216},"labels":[],"label_agreement":null},{"id":"W2072764341","doi":"10.1080/09695940600563611","title":"Differential effects of global modifications to large‐scale high stakes examination programmes","year":2006,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; Queen's University","funders":"","keywords":"Scholarship; Accountability; Scale (ratio); Session (web analytics); Test (biology); Empirical examination; Identification (biology); Psychology; Political science; Medical education; Medicine; Geography; Law; Actuarial science; Business; Advertising","score_opus":0.02697556704776101,"score_gpt":0.41688929703418975,"score_spread":0.38991372998642876,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2072764341","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9905674,0.00027148725,0.00065994717,0.0005236309,0.00009257206,0.0002329693,0.00015488954,0.00010919885,0.007387961],"genre_scores_gemma":[0.996357,0.000100101126,0.00067292363,0.00023870674,0.00003626014,0.00010449759,0.00011856779,0.0000397402,0.0023321856],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98720586,0.006351449,0.00092829915,0.0014912084,0.0025000253,0.0015231473],"domain_scores_gemma":[0.9452789,0.03262029,0.0071814563,0.0065671536,0.0033437875,0.0050085257],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008889036,0.0006268205,0.00087824237,0.001015656,0.0007086614,0.0018573365,0.0012813318,0.0010813521,0.009779217],"category_scores_gemma":[0.07017854,0.00031757326,0.00087923213,0.001295103,0.0020312094,0.0013574059,0.0040929434,0.0012944186,0.0008612818],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.019845316,0.008258079,0.32675573,0.0012312583,0.0011573995,0.0013797089,0.008124097,0.015740296,0.029159168,0.0052903136,0.0043976163,0.5786612],"study_design_scores_gemma":[0.0003380403,0.01041602,0.9727098,0.00013828657,0.00027367167,0.00017459168,0.0025893876,0.0011938802,0.005266742,0.0010610058,0.005779342,0.000059150265],"about_ca_topic_score_codex":0.003920957,"about_ca_topic_score_gemma":0.0072783376,"teacher_disagreement_score":0.009779217,"about_ca_system_score_codex":0.0021736068,"about_ca_system_score_gemma":0.0012492482,"threshold_uncertainty_score":0.047010243},"labels":[],"label_agreement":null},{"id":"W2124002552","doi":"10.1080/0969594x.2014.967168","title":"Instructional Rounds as a professional learning model for systemic implementation of Assessment for Learning","year":2014,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":52,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Queen's University","funders":"","keywords":"Professional learning community; Professional development; Psychology; Value (mathematics); Pedagogy; Mathematics education; Session (web analytics); Medical education; Medicine; Computer science","score_opus":0.058823989514014775,"score_gpt":0.5102705274787012,"score_spread":0.45144653796468637,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2124002552","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.23667203,0.0005523081,0.47967136,0.017342234,0.00020435324,0.00394096,0.000091902584,0.0011005884,0.26042435],"genre_scores_gemma":[0.8017721,0.00020082905,0.18455575,0.0005642489,0.000027053004,0.0011399777,0.000038010377,0.000057641402,0.01164435],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.95770925,0.033728816,0.00089190016,0.0015697442,0.004691265,0.0014089942],"domain_scores_gemma":[0.9696437,0.014655311,0.0028472417,0.0046707257,0.005223579,0.0029595057],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.023705699,0.0004270737,0.00023211294,0.0011358649,0.002888883,0.005527039,0.0018150407,0.00089778745,0.0031825032],"category_scores_gemma":[0.02983303,0.00049740664,0.00035759207,0.00068775704,0.007585511,0.003834024,0.0044706226,0.0022351854,0.0008174926],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018421977,0.0016521158,0.03238274,0.00093822833,0.000044196007,0.0005284951,0.11242345,0.009633849,0.005130688,0.40580192,0.008707254,0.42257282],"study_design_scores_gemma":[0.0004405592,0.0047594467,0.057692725,0.0017869486,0.000114737064,0.0018018124,0.09455261,0.0730367,0.011325941,0.28623784,0.46786663,0.00038408232],"about_ca_topic_score_codex":0.013454142,"about_ca_topic_score_gemma":0.033154868,"teacher_disagreement_score":0.023705699,"about_ca_system_score_codex":0.010910666,"about_ca_system_score_gemma":0.029850464,"threshold_uncertainty_score":0.12536925},"labels":[],"label_agreement":null},{"id":"W2287186680","doi":"10.1080/0969594x.2015.1113929","title":"Brazilian national assessment data and educational policy: an empirical illustration","year":2015,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"School Choice and Performance","field":"Social Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"","keywords":"Portuguese; Sample (material); Census; Latin Americans; Student achievement; Mathematics education; Geography; Developing country; Contrast (vision); Academic achievement; Political science; Psychology; Economic growth; Demography; Sociology; Population; Computer science; Economics","score_opus":0.23858880022483428,"score_gpt":0.5617100844317808,"score_spread":0.3231212842069465,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2287186680","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.60964817,0.018404063,0.035313483,0.039259728,0.000333932,0.0009997193,0.017762773,0.0002902245,0.27798787],"genre_scores_gemma":[0.9810153,0.0026714534,0.011998268,0.0004909793,0.0000634363,0.00029296026,0.002306371,0.00004137961,0.001119821],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9757237,0.014908125,0.0015448172,0.0014128358,0.0054524066,0.0009581333],"domain_scores_gemma":[0.81996256,0.13765492,0.011089062,0.0093366485,0.02061932,0.0013376159],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.026583765,0.00040141924,0.0006354437,0.006483433,0.0017532402,0.0028494408,0.0011914774,0.0010931165,0.00585203],"category_scores_gemma":[0.14301306,0.00039924373,0.00065993593,0.026661428,0.0026933416,0.0032886348,0.002752471,0.0013562706,0.000493741],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008781666,0.00031176017,0.6665403,0.0009821999,0.00016571202,0.00051452254,0.0067786467,0.002979108,0.00013144677,0.19340368,0.01674501,0.1113598],"study_design_scores_gemma":[0.00008655252,0.00022204552,0.70075405,0.004150238,0.0003924017,0.0009795412,0.028465146,0.039742727,0.00052401895,0.074453846,0.15011013,0.00011941753],"about_ca_topic_score_codex":0.16643836,"about_ca_topic_score_gemma":0.14025988,"teacher_disagreement_score":0.16643836,"about_ca_system_score_codex":0.005862864,"about_ca_system_score_gemma":0.009602606,"threshold_uncertainty_score":0.3309391},"labels":[],"label_agreement":null},{"id":"W2317555153","doi":"10.1080/0969594x.2013.844094","title":"Students’ interpersonal trust and attitudes towards standardised tests: exploring affective variables related to student assessment","year":2013,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Communication in Education and Healthcare","field":"Psychology","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Psychology; Interpersonal communication; Social psychology; Test (biology); Expectancy theory; Scale (ratio); Value (mathematics); Structural equation modeling; Interpersonal relationship; Computer science","score_opus":0.09861461347341459,"score_gpt":0.526713327331136,"score_spread":0.42809871385772136,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2317555153","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9991891,0.0000208559,0.00028747306,0.000041954936,0.0000017725567,0.0000045853712,0.0000063810858,0.000001404688,0.00044648323],"genre_scores_gemma":[0.99974126,0.000018684525,0.00011773662,0.000012063083,0.0000027954352,0.000004598585,0.000010287565,8.5315986e-7,0.000091741225],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9958276,0.0021144669,0.00032939194,0.00019474293,0.0012325986,0.00030120264],"domain_scores_gemma":[0.9398021,0.032899886,0.018208548,0.0023565153,0.0033462406,0.0033867492],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0060081896,0.00025564787,0.00035757571,0.00090406113,0.0004625218,0.002124954,0.00031489786,0.00048270824,0.0018131118],"category_scores_gemma":[0.045755222,0.0002100104,0.00041081998,0.0008222148,0.0009834606,0.00084174954,0.0011128464,0.0012473952,0.00024868705],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007688852,0.0003966136,0.9858148,0.00001826393,0.00006125593,0.00004308258,0.0030560922,0.00020273218,0.00046404597,0.00014855787,0.00005556912,0.009662081],"study_design_scores_gemma":[0.000003582776,0.00025222238,0.99518025,0.000013890169,0.000020226678,0.000054319393,0.002237995,0.0012919778,0.0005142628,0.00026613867,0.000150521,0.0000145350905],"about_ca_topic_score_codex":0.0021700477,"about_ca_topic_score_gemma":0.0032096591,"teacher_disagreement_score":0.0060081896,"about_ca_system_score_codex":0.0006653706,"about_ca_system_score_gemma":0.00065739895,"threshold_uncertainty_score":0.03177476},"labels":[],"label_agreement":null},{"id":"W2600733185","doi":"10.1080/0969594x.2017.1297010","title":"Developing assessment capable teachers in this age of accountability","year":2017,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":75,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Accountability; Psychology; Medical education; Political science; Medicine","score_opus":0.12258459740226811,"score_gpt":0.5148076355503635,"score_spread":0.39222303814809534,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2600733185","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.11047169,0.0077255713,0.09586354,0.51935446,0.0043922113,0.00032203618,0.00011926294,0.0013457026,0.26040554],"genre_scores_gemma":[0.7978379,0.006200644,0.051717814,0.041979495,0.0011525952,0.00051115284,0.00015188458,0.00040900335,0.100039475],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.98399323,0.008023634,0.0006391404,0.0013605729,0.0034255872,0.0025579212],"domain_scores_gemma":[0.95298755,0.016498078,0.0041022003,0.003371358,0.009192582,0.013848302],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02269946,0.00037476222,0.00051519915,0.00089560624,0.006982223,0.013473852,0.0016558691,0.0062753805,0.0093769375],"category_scores_gemma":[0.04361229,0.00055352703,0.00024319878,0.00065627805,0.007994081,0.018247254,0.016736776,0.010582175,0.005695342],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001354123,0.0005840354,0.015658066,0.00068845076,0.000023333941,0.0011800148,0.10901565,0.0010800326,0.0032269717,0.44078654,0.13930124,0.28832027],"study_design_scores_gemma":[0.00003705346,0.00016258963,0.0064823152,0.0009940208,0.000012044042,0.00087768224,0.060138598,0.0017207453,0.0020065568,0.18313481,0.74435776,0.00007589757],"about_ca_topic_score_codex":0.0034595646,"about_ca_topic_score_gemma":0.007254483,"teacher_disagreement_score":0.02269946,"about_ca_system_score_codex":0.0041013625,"about_ca_system_score_gemma":0.020156164,"threshold_uncertainty_score":0.12004763},"labels":[],"label_agreement":null},{"id":"W2923740196","doi":"10.1080/0969594x.2019.1593105","title":"Conceptualising fairness in classroom assessment: exploring the value of organisational justice theory","year":2019,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":71,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Foundation (evidence); Economic Justice; Scholarship; Value (mathematics); Core (optical fiber); Sociology; Psychology; Social psychology; Epistemology; Political science; Computer science; Law","score_opus":0.07905727411628827,"score_gpt":0.4421515498552321,"score_spread":0.3630942757389438,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2923740196","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17688383,0.014725224,0.56537503,0.0845071,0.0010217563,0.00040189983,0.000051896513,0.00011234197,0.15692092],"genre_scores_gemma":[0.9659733,0.0012639746,0.03027358,0.0011333204,0.00016372815,0.00015038559,0.00000934343,0.000021164218,0.0010111178],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.92918247,0.057778794,0.0016877895,0.0018974914,0.007637205,0.0018163425],"domain_scores_gemma":[0.8795193,0.10270623,0.0052090506,0.004364857,0.0058338298,0.0023667829],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.052763622,0.00056488416,0.0010955626,0.00401221,0.005696093,0.014909508,0.00298669,0.004119467,0.0025402678],"category_scores_gemma":[0.09571839,0.00039292817,0.00071360084,0.002635494,0.03781327,0.016124867,0.013678789,0.0077295196,0.00022301327],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000026291356,0.00011629466,0.0038825825,0.00022937491,0.000019363431,0.000075961936,0.036736887,0.0017251553,0.00014906038,0.9147874,0.00059576856,0.04165575],"study_design_scores_gemma":[0.000012766865,0.000047639984,0.001803405,0.00067059445,0.000013903389,0.0000785082,0.015513748,0.0049208663,0.00021482188,0.9650324,0.011660699,0.000030673473],"about_ca_topic_score_codex":0.0058271484,"about_ca_topic_score_gemma":0.0055174134,"teacher_disagreement_score":0.052763622,"about_ca_system_score_codex":0.010135305,"about_ca_system_score_gemma":0.013371337,"threshold_uncertainty_score":0.27904403},"labels":[],"label_agreement":null},{"id":"W2994689616","doi":"10.1080/0969594x.2019.1703171","title":"A cross-cultural comparison of German and Canadian student teachers’ assessment competence","year":2019,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":26,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Queen's University","funders":"","keywords":"Competence (human resources); German; Multitude; Psychology; Pedagogy; Cross-cultural; Mathematics education; Sociology; Social psychology; Political science; Geography","score_opus":0.06255569384159228,"score_gpt":0.5279710097896466,"score_spread":0.46541531594805435,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2994689616","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99280703,0.00027185443,0.00015073783,0.00013063897,0.000013120425,0.000017677936,0.00014188288,0.0000037482469,0.006463303],"genre_scores_gemma":[0.99841475,0.00019810884,0.00017370845,0.000042267027,0.0000018171064,0.000010292882,0.00014982033,0.000004014094,0.0010052266],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99690837,0.00046759428,0.00017963986,0.0003015148,0.0014424155,0.00070049905],"domain_scores_gemma":[0.99257267,0.0015029414,0.0005483448,0.0002849989,0.004225052,0.00086589734],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036476415,0.00024626948,0.00038557168,0.003032211,0.0038397778,0.0021293468,0.0005950579,0.0003147577,0.0020390674],"category_scores_gemma":[0.008860679,0.00022993864,0.00034235528,0.0044149775,0.002149972,0.00059573003,0.0015865926,0.00062334194,0.00018079676],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00045397386,0.00021871523,0.54853684,0.0002801712,0.00014343846,0.00052698434,0.36012077,0.00033061617,0.0033230956,0.004211653,0.003057968,0.07879571],"study_design_scores_gemma":[0.0000096825515,0.00010557955,0.8485903,0.000101473,0.00003359132,0.00018783036,0.14220797,0.00018401672,0.00060828374,0.00013921091,0.0077843396,0.000047702844],"about_ca_topic_score_codex":0.95181054,"about_ca_topic_score_gemma":0.9812887,"teacher_disagreement_score":0.04818946,"about_ca_system_score_codex":0.016998116,"about_ca_system_score_gemma":0.018107701,"threshold_uncertainty_score":0.12333053},"labels":[],"label_agreement":null},{"id":"W3004711011","doi":"10.1080/0969594x.2020.1719033","title":"Leveraging assessment to promote kindergarten learners’ independence and self-regulation within play-based classrooms","year":2020,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Early Childhood Education and Development","field":"Social Sciences","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Queen's University","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Curriculum; Leverage (statistics); Psychology; Pedagogy; Early childhood education; Independence (probability theory); Assessment for learning; Mathematics education; Formative assessment; Computer science","score_opus":0.048417935656527564,"score_gpt":0.3940303882998441,"score_spread":0.3456124526433165,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3004711011","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.92397374,0.0008223134,0.020549593,0.0023229725,0.000039152914,0.0001823387,0.000017769025,0.00021782705,0.05187421],"genre_scores_gemma":[0.9881442,0.0003081306,0.009693552,0.00008900546,0.0000047892904,0.000067473666,0.000010591208,0.000011366869,0.001670825],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9941017,0.002955224,0.00037284175,0.00062480103,0.0013020508,0.0006434683],"domain_scores_gemma":[0.9927443,0.0038154763,0.00094492227,0.0006692867,0.0008338373,0.0009920868],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006664889,0.00031886232,0.00038351564,0.0014131098,0.0018509332,0.004719397,0.0009890444,0.00044277633,0.0008595724],"category_scores_gemma":[0.012012584,0.00028455295,0.00019720646,0.0007168405,0.0041622007,0.001713164,0.0072943517,0.0014251466,0.00023929412],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00008638376,0.0012909104,0.13588381,0.0005946955,0.00004337688,0.00096159615,0.27462718,0.0015984726,0.01675508,0.020254511,0.0019335544,0.5459704],"study_design_scores_gemma":[0.00007761817,0.0012639669,0.5029361,0.0027863004,0.00014863309,0.0017192949,0.26663586,0.004381692,0.030895902,0.044398177,0.14449157,0.00026490193],"about_ca_topic_score_codex":0.007137745,"about_ca_topic_score_gemma":0.021320747,"teacher_disagreement_score":0.007137745,"about_ca_system_score_codex":0.0022608226,"about_ca_system_score_gemma":0.0059923264,"threshold_uncertainty_score":0.035247684},"labels":[],"label_agreement":null},{"id":"W3007758995","doi":"10.1080/0969594x.2020.1728908","title":"Making feedback effective?","year":2020,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kellogg's (Canada)","funders":"","keywords":"Computer science; Psychology","score_opus":0.09632089257636077,"score_gpt":0.48106678362533445,"score_spread":0.38474589104897366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3007758995","genre_codex":"commentary","genre_gemma":"commentary","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"commentary","genre_consensus":"commentary","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.026804011,0.024142802,0.040364005,0.7860187,0.012691897,0.00055831706,0.00017940467,0.0015422738,0.10769851],"genre_scores_gemma":[0.7565559,0.02505055,0.075075045,0.11118805,0.007897443,0.0013709604,0.00019375549,0.00097046315,0.021697806],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.86509293,0.08764392,0.0057533374,0.004702753,0.031002143,0.005804846],"domain_scores_gemma":[0.7039152,0.18336879,0.020660916,0.014270932,0.06266631,0.015117802],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.096601814,0.0010701199,0.0011396463,0.0022819454,0.0039629363,0.012845359,0.002000519,0.007339032,0.01456862],"category_scores_gemma":[0.38123378,0.00072602835,0.00092754577,0.0011638846,0.006621133,0.018497882,0.005901547,0.0066248192,0.0059613143],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023044366,0.00036356744,0.006821586,0.0023375354,0.00016000959,0.00026450184,0.016911557,0.00023401417,0.0012616925,0.03923567,0.17929852,0.7528809],"study_design_scores_gemma":[0.0005023488,0.0008901375,0.013348445,0.01696025,0.0004385887,0.0017501421,0.042202104,0.0020406593,0.0051033245,0.15742135,0.7591079,0.00023470905],"about_ca_topic_score_codex":0.0025421858,"about_ca_topic_score_gemma":0.0033015262,"teacher_disagreement_score":0.096601814,"about_ca_system_score_codex":0.004683778,"about_ca_system_score_gemma":0.013941105,"threshold_uncertainty_score":0.51088536},"labels":[],"label_agreement":null},{"id":"W3081330604","doi":"10.1080/0969594x.2020.1801576","title":"Collaborating with teachers to design and implement assessments for self-regulated learning in the context of authentic classroom writing tasks","year":2020,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Innovative Teaching and Learning Methods","field":"Psychology","cited_by":41,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; University of British Columbia","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Context (archaeology); Task (project management); Mathematics education; Psychology; Self-regulated learning; Quality (philosophy); Pedagogy; Computer science; Engineering","score_opus":0.10441676341013247,"score_gpt":0.49469630880821713,"score_spread":0.3902795453980847,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3081330604","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.59202814,0.0014444456,0.37197617,0.004151878,0.00040743483,0.00888808,0.0005494108,0.0014352954,0.019119194],"genre_scores_gemma":[0.5936783,0.0003836199,0.38903072,0.0007120684,0.000040102488,0.0077588493,0.00032108388,0.00024277616,0.007832447],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9423909,0.03988245,0.00344763,0.0069825766,0.005508814,0.0017876435],"domain_scores_gemma":[0.8415373,0.07077259,0.017820453,0.025692714,0.035994105,0.008182816],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07075043,0.00058326655,0.0008026394,0.0028615054,0.003988469,0.006416522,0.0021754599,0.0011143533,0.003341679],"category_scores_gemma":[0.119812936,0.0011299961,0.00072380865,0.0016371438,0.00228313,0.003514174,0.0052415235,0.0036430194,0.0020587523],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027132803,0.003107663,0.15728563,0.0009839134,0.000111602,0.00042556837,0.30276474,0.001426001,0.01773218,0.004439943,0.005949888,0.50550157],"study_design_scores_gemma":[0.0008977152,0.004567588,0.20383668,0.003116529,0.00036702707,0.0019045563,0.36466318,0.013228533,0.059335355,0.02314136,0.3243857,0.0005558423],"about_ca_topic_score_codex":0.00416108,"about_ca_topic_score_gemma":0.012287757,"teacher_disagreement_score":0.07075043,"about_ca_system_score_codex":0.0037250482,"about_ca_system_score_gemma":0.018457929,"threshold_uncertainty_score":0.37416852},"labels":[],"label_agreement":null},{"id":"W3167522218","doi":"10.1080/0969594x.2021.1932736","title":"Conceptualising a Fairness Framework for Assessment Adjusted Practices for Students with Disability: An Empirical Study","year":2021,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Disability Education and Employment","field":"Social Sciences","cited_by":21,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Leverage (statistics); Psychology; Diversity (politics); Best practice; Pedagogy; Medical education; Mathematics education; Applied psychology; Sociology; Medicine; Computer science","score_opus":0.24808421599986105,"score_gpt":0.6006909723976173,"score_spread":0.3526067563977562,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3167522218","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95312536,0.0006461915,0.025463047,0.006592542,0.000042035346,0.00036466756,0.00001772799,0.000017297316,0.0137311695],"genre_scores_gemma":[0.99478525,0.00015861819,0.0045662834,0.00015811135,0.0000056638696,0.00009470297,0.00000390247,0.0000033665629,0.00022418938],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9461273,0.04086717,0.0022501461,0.0016836359,0.0068788477,0.0021928481],"domain_scores_gemma":[0.9252756,0.050874434,0.009980464,0.0034472393,0.006033781,0.004388401],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05534475,0.00038883876,0.0006902994,0.0029389665,0.010221407,0.009255775,0.0026803303,0.002080311,0.0012532144],"category_scores_gemma":[0.06347857,0.00051780354,0.00048629552,0.0025789938,0.023530588,0.008916869,0.012515863,0.005055319,0.00010744337],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00002097176,0.00034319292,0.03994433,0.00017132242,0.000012491286,0.00044902548,0.8651881,0.00046672343,0.00043190634,0.05976109,0.0004251888,0.032785643],"study_design_scores_gemma":[0.000015073813,0.00011416459,0.027991008,0.00078608165,0.000017522872,0.0004983243,0.9155991,0.0032033704,0.0003790321,0.035714418,0.015622541,0.000059466984],"about_ca_topic_score_codex":0.008281191,"about_ca_topic_score_gemma":0.010010971,"teacher_disagreement_score":0.05534475,"about_ca_system_score_codex":0.010897001,"about_ca_system_score_gemma":0.01962489,"threshold_uncertainty_score":0.2926945},"labels":[],"label_agreement":null},{"id":"W3185953117","doi":"10.1080/0969594x.2021.1951161","title":"Assessing young children’s self-regulation in school contexts","year":2021,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Innovative Teaching and Learning Methods","field":"Psychology","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; The King's University; Western University","funders":"","keywords":"Self-regulated learning; Psychology; Curriculum; Developmental psychology; Situated; Self-control; Pedagogy; Computer science","score_opus":0.05781325932187747,"score_gpt":0.4946531875119819,"score_spread":0.4368399281901044,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3185953117","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9960835,0.0001925219,0.0006900243,0.000030299834,0.0000049258256,0.000071937226,0.0001532242,0.000020568576,0.0027529837],"genre_scores_gemma":[0.99426794,0.00053208234,0.0033487119,0.000023265968,0.000005345551,0.00022374268,0.00044135685,0.000009345943,0.001148137],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9989926,0.00027910867,0.0001630668,0.00014823292,0.00025806145,0.00015897614],"domain_scores_gemma":[0.99787414,0.0006237456,0.000757989,0.00018979512,0.0003594202,0.00019501198],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020174014,0.00026051808,0.0004095766,0.0008915102,0.00038376727,0.00086362305,0.00032879048,0.00024982847,0.0010492796],"category_scores_gemma":[0.0037995095,0.00021152766,0.0003974646,0.00067916454,0.00039570322,0.00045778282,0.00078667095,0.0006072127,0.00026603034],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007423351,0.00018278562,0.9270488,0.00016729753,0.00005078199,0.00014576846,0.012671542,0.00025913824,0.0037083535,0.00033588687,0.00043546615,0.054919966],"study_design_scores_gemma":[0.000002796899,0.000118399716,0.9950547,0.000026765494,0.000012065649,0.00011899321,0.0024241116,0.00007238091,0.0010416835,0.000109994486,0.0010091953,0.000008858147],"about_ca_topic_score_codex":0.0035321787,"about_ca_topic_score_gemma":0.0071740556,"teacher_disagreement_score":0.0035321787,"about_ca_system_score_codex":0.00044678594,"about_ca_system_score_gemma":0.0005537207,"threshold_uncertainty_score":0.010669231},"labels":[],"label_agreement":null},{"id":"W3210068963","doi":"10.1080/0969594x.2021.1988510","title":"Formative assessment, growth mindset, and achievement: examining their relations in the East and the West","year":2021,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":50,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"","funders":"","keywords":"Mindset; Formative assessment; Psychology; Reading (process); Mainland China; Academic achievement; Pedagogy; Mathematics education; China; Political science; Computer science","score_opus":0.07796939436320567,"score_gpt":0.42193192942447144,"score_spread":0.3439625350612658,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3210068963","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99925095,0.00006910033,0.00010484408,0.00002123917,0.0000014200401,0.0000017002203,0.000026914457,0.0000011277203,0.0005227565],"genre_scores_gemma":[0.9996458,0.000059283215,0.00014038017,0.000006732942,0.0000017708203,0.0000022403012,0.00003035089,0.0000012846076,0.00011203233],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9994117,0.00019692093,0.000083034945,0.00009223673,0.00013966621,0.00007642159],"domain_scores_gemma":[0.9946791,0.0016212311,0.0020709813,0.0003556135,0.00077909423,0.0004939828],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014838835,0.00027666756,0.0004068755,0.0014838107,0.00033836195,0.0012583241,0.000190002,0.00015440269,0.00055812416],"category_scores_gemma":[0.0038120304,0.00015686669,0.00029478958,0.0015439271,0.00059819565,0.00075291615,0.0010268198,0.00042341326,0.00011061561],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00006492984,0.000033590262,0.9881188,0.000016439708,0.000050575847,0.000043020988,0.002015837,0.000066746434,0.0006681212,0.00013025814,0.000021244678,0.008770436],"study_design_scores_gemma":[0.0000016179123,0.000045835033,0.99778855,0.0000102192325,0.000021285125,0.00003132317,0.0013606809,0.00026634548,0.0002881409,0.000059493963,0.00012321229,0.0000032241237],"about_ca_topic_score_codex":0.013181362,"about_ca_topic_score_gemma":0.02249004,"teacher_disagreement_score":0.013181362,"about_ca_system_score_codex":0.00039762395,"about_ca_system_score_gemma":0.0004951839,"threshold_uncertainty_score":0.026209295},"labels":[],"label_agreement":null},{"id":"W4205974171","doi":"10.1080/0969594x.2021.1999209","title":"Investigating the potential of NLP-driven linguistic and acoustic features for predicting human scores of children’s oral language proficiency","year":2021,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Reading and Literacy Development","field":"Psychology","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Storytelling; Vocabulary; Grammar; Literacy; Psychology; Computer science; Linguistics; Natural language processing; Artificial intelligence; Narrative; Pedagogy","score_opus":0.030720207226523362,"score_gpt":0.42304993140413255,"score_spread":0.3923297241776092,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4205974171","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9932673,0.00006910231,0.004340968,0.0000281821,0.000004894812,0.000018606564,0.0002773609,0.000038450074,0.001955219],"genre_scores_gemma":[0.99520266,0.000042086747,0.0042287875,0.000007924007,0.0000033336837,0.000025615378,0.00021263973,0.000008314098,0.00026856246],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99856466,0.0006664735,0.000100578894,0.0003155547,0.00027771917,0.00007500731],"domain_scores_gemma":[0.98545444,0.010571166,0.0018083235,0.00083349645,0.0010506781,0.00028185514],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025696652,0.00043721937,0.00017621608,0.0010953086,0.00014582835,0.00092437107,0.00025781977,0.0003226456,0.0014502489],"category_scores_gemma":[0.0142540885,0.00018585572,0.00022807742,0.0005280212,0.00042485,0.00089185854,0.00067091454,0.0004014818,0.00050527736],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020803025,0.00010453079,0.9256448,0.00007718473,0.00006916486,0.00011066477,0.0017155608,0.0018153436,0.009085123,0.0003311131,0.00019965133,0.060638845],"study_design_scores_gemma":[0.0000074342156,0.00032998386,0.9814379,0.000025807367,0.000026444022,0.0002609379,0.0011086577,0.009700258,0.006009785,0.00039075228,0.0006737213,0.000028372679],"about_ca_topic_score_codex":0.002729662,"about_ca_topic_score_gemma":0.005503613,"teacher_disagreement_score":0.002729662,"about_ca_system_score_codex":0.00019296168,"about_ca_system_score_gemma":0.00036070557,"threshold_uncertainty_score":0.013589799},"labels":[],"label_agreement":null},{"id":"W4286809449","doi":"10.1080/0969594x.2022.2103516","title":"The Education and Assessment System in Lithuania","year":2022,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Higher Education Learning Practices","field":"Social Sciences","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Declaration; Accountability; Political science; Curriculum; Bologna declaration; Politics; Test (biology); Public administration; Independence (probability theory); Function (biology); European union; Educational assessment; Pedagogy; Public relations; Higher education; Sociology; Business; Law","score_opus":0.04449651455830835,"score_gpt":0.47242111135392556,"score_spread":0.42792459679561723,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4286809449","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.67246,0.02687326,0.004489002,0.028647717,0.00071440823,0.00014273456,0.00060859637,0.000417261,0.26564696],"genre_scores_gemma":[0.9804027,0.0018327627,0.0011968564,0.0012752261,0.000085638894,0.000044463075,0.00014995589,0.000027853725,0.014984437],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","domain_scores_codex":[0.99530053,0.001293331,0.00053971633,0.00048299134,0.00069271575,0.0016907419],"domain_scores_gemma":[0.99856395,0.00032998377,0.00036068502,0.00014903962,0.00039898654,0.00019739817],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032219526,0.00032018835,0.00039209146,0.0028296679,0.0035757138,0.008160694,0.0009558166,0.0016462565,0.0033714462],"category_scores_gemma":[0.0025991697,0.00033846422,0.00031187545,0.0032601259,0.0064078183,0.001880236,0.0060090413,0.0013629048,0.00071442826],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028831526,0.00014475503,0.07863945,0.001187772,0.000098982746,0.005312048,0.039062608,0.004644328,0.002162377,0.57687044,0.015785994,0.275803],"study_design_scores_gemma":[0.00007148285,0.0003096188,0.22069927,0.002119994,0.000060634557,0.0025427728,0.021098575,0.0015942622,0.0030207867,0.034250993,0.71403563,0.00019592908],"about_ca_topic_score_codex":0.032212157,"about_ca_topic_score_gemma":0.018253334,"teacher_disagreement_score":0.032212157,"about_ca_system_score_codex":0.013155579,"about_ca_system_score_gemma":0.021824317,"threshold_uncertainty_score":0.09545088},"labels":[],"label_agreement":null},{"id":"W4322503168","doi":"10.1080/0969594x.2023.2182737","title":"Data literacy assessments: a systematic literature review","year":2023,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Educational Assessment and Improvement","field":"Decision Sciences","cited_by":57,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Literacy; Reliability (semiconductor); Computer science; Information literacy; Field (mathematics); Quality (philosophy); Systematic review; Psychology; Medical education; Data science; Mathematics education; Pedagogy; Political science; Medicine; MEDLINE","score_opus":0.25638355098970766,"score_gpt":0.588095050840592,"score_spread":0.33171149985088433,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4322503168","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0012671823,0.99298567,0.0009899064,0.0010116607,0.00024534896,0.0019632082,0.0007764909,0.000028726176,0.00073173834],"genre_scores_gemma":[0.012960115,0.978225,0.003908332,0.0010156279,0.00011497892,0.0031256715,0.00044596838,0.000012980047,0.000191234],"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","domain_scores_codex":[0.9660907,0.013332933,0.011569098,0.0016565544,0.006780194,0.0005704663],"domain_scores_gemma":[0.8707806,0.096491575,0.015049482,0.0021649946,0.014341176,0.0011722543],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.031843327,0.0018443099,0.0071050255,0.028107973,0.0015168526,0.0038596382,0.0025659062,0.0025492252,0.0051024174],"category_scores_gemma":[0.14512116,0.0014951762,0.0070637423,0.019860594,0.0018521221,0.0055073984,0.0037025956,0.0020399129,0.0007074404],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000104410756,0.000035716923,0.00065233983,0.9159872,0.001990832,0.00012235653,0.00048253682,0.00012155265,0.00013460811,0.0004264448,0.0030003958,0.07694165],"study_design_scores_gemma":[0.000073409305,0.00007809194,0.0013226137,0.97331434,0.007678452,0.00018502996,0.0003916472,0.00006085526,0.0000991843,0.00034892219,0.01642221,0.000025290488],"about_ca_topic_score_codex":0.008263989,"about_ca_topic_score_gemma":0.026467556,"teacher_disagreement_score":0.9681567,"about_ca_system_score_codex":0.0065168566,"about_ca_system_score_gemma":0.03785942,"threshold_uncertainty_score":0.16840559},"labels":[],"label_agreement":null},{"id":"W4385416815","doi":"10.1080/0969594x.2023.2242004","title":"Educational assessment in Ghana: The influence of historical colonization and political accountability","year":2023,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Formative assessment; Summative assessment; Accountability; Context (archaeology); Politics; Political science; Educational assessment; Government (linguistics); Pedagogy; Sociology; Public administration; Geography; Law","score_opus":0.05672128565655436,"score_gpt":0.46713834296963386,"score_spread":0.4104170573130795,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385416815","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.76382256,0.02373879,0.002272203,0.09861409,0.00043402394,0.00007995665,0.000107942455,0.000034095872,0.110896304],"genre_scores_gemma":[0.9936592,0.0033063802,0.00035551545,0.0008955793,0.000053992782,0.000010288511,0.000010322761,0.000009955346,0.0016988206],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.9968214,0.0020697175,0.00012626455,0.00022349221,0.00031269854,0.00044647255],"domain_scores_gemma":[0.9899032,0.0057315347,0.0019004018,0.00028884696,0.0011518673,0.0010242257],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005369835,0.00013378053,0.00017037998,0.0013055588,0.003322731,0.003736515,0.00038374055,0.00071763573,0.0026599385],"category_scores_gemma":[0.011900704,0.00022643803,0.00007192354,0.0025529843,0.0105128735,0.0037658755,0.002478466,0.0019907884,0.00020236894],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018497913,0.00015253831,0.09148297,0.0008859916,0.0000140175725,0.007195457,0.46750125,0.0005140936,0.0016820063,0.1900838,0.010982916,0.22931993],"study_design_scores_gemma":[0.000030253928,0.00015603262,0.23545821,0.0031494743,0.000018584486,0.0030300766,0.26520082,0.0006623362,0.0012061028,0.028425762,0.4625871,0.00007531348],"about_ca_topic_score_codex":0.03482587,"about_ca_topic_score_gemma":0.05989663,"teacher_disagreement_score":0.03482587,"about_ca_system_score_codex":0.010398418,"about_ca_system_score_gemma":0.0073672584,"threshold_uncertainty_score":0.07544613},"labels":[],"label_agreement":null},{"id":"W4386496830","doi":"10.1080/0969594x.2023.2255936","title":"Classroom assessment fairness inventory: a new instrument to support perceived fairness in classroom assessment","year":2023,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":true,"ca_institutions":"Queen's University; University of Saskatchewan; University of Alberta","funders":"","keywords":"Psychology; Grading (engineering); Equity (law); Perception; Legitimacy; Diversity (politics); Pedagogy; Medical education; Sociology; Political science","score_opus":0.08886918950009094,"score_gpt":0.4589116453380946,"score_spread":0.37004245583800366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386496830","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8206001,0.00079088897,0.07308629,0.0015526628,0.00034575426,0.00795811,0.004821991,0.0009084716,0.08993581],"genre_scores_gemma":[0.8504209,0.000755763,0.12849775,0.0003143583,0.00008831438,0.00577275,0.0024200405,0.000113639224,0.011616477],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9902368,0.0021437192,0.001467936,0.00049545226,0.005106527,0.0005496078],"domain_scores_gemma":[0.9597449,0.01256975,0.0064809034,0.00328822,0.015178542,0.002737746],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012783476,0.0004200992,0.00068040716,0.0033437696,0.0019774523,0.0020984216,0.0011702926,0.0004282756,0.0036697763],"category_scores_gemma":[0.033985958,0.00034266745,0.00088212546,0.0020315216,0.001341013,0.0016968456,0.0027926539,0.0016679665,0.0007078187],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047681152,0.0018646744,0.45401898,0.0004408814,0.00018481405,0.00017182002,0.016738342,0.0016487817,0.0058285883,0.007685308,0.01632489,0.4946162],"study_design_scores_gemma":[0.00013059264,0.00053515466,0.93632066,0.00034854776,0.00010384048,0.00018172123,0.005942031,0.0049433997,0.0029052643,0.0058765495,0.042483136,0.00022926087],"about_ca_topic_score_codex":0.090728894,"about_ca_topic_score_gemma":0.22003035,"teacher_disagreement_score":0.090728894,"about_ca_system_score_codex":0.006748262,"about_ca_system_score_gemma":0.013734606,"threshold_uncertainty_score":0.18040156},"labels":[],"label_agreement":null},{"id":"W4411558571","doi":"10.1080/0969594x.2025.2523162","title":"Placing students at the centre: new directions for student agency in assessment","year":2025,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Higher Education Learning Practices","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Agency (philosophy); Mathematics education; Psychology; Medical education; Pedagogy; Sociology; Medicine; Social science","score_opus":0.056306342601075456,"score_gpt":0.535095557930777,"score_spread":0.47878921532970153,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411558571","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0011243294,0.01106158,0.010753246,0.9449308,0.0082320245,0.000028640992,0.000007910629,0.00018950904,0.02367208],"genre_scores_gemma":[0.415751,0.05506526,0.09457074,0.31013045,0.045091998,0.0010193143,0.00009302495,0.0015430965,0.07673506],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.84548825,0.12093394,0.006067643,0.0053182016,0.016250806,0.00594118],"domain_scores_gemma":[0.7945149,0.1312061,0.0056225527,0.016636278,0.026207354,0.025812857],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.100065,0.0007005335,0.0014671938,0.0020723192,0.015486183,0.044912595,0.005275916,0.021524237,0.015918862],"category_scores_gemma":[0.12837996,0.0008726469,0.0015221171,0.0020213549,0.053339634,0.061696533,0.026669597,0.033252787,0.004679092],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007538074,0.00015274635,0.0016390135,0.0005573783,0.000025146941,0.000121754165,0.03599612,0.00023200088,0.00010491712,0.44055125,0.2977837,0.22276066],"study_design_scores_gemma":[0.000045312478,0.00009105981,0.000792863,0.001271332,0.00001650528,0.0001979109,0.021651583,0.00043160922,0.00021921202,0.20445934,0.7707296,0.00009364604],"about_ca_topic_score_codex":0.0060816365,"about_ca_topic_score_gemma":0.013881176,"teacher_disagreement_score":0.100065,"about_ca_system_score_codex":0.012841085,"about_ca_system_score_gemma":0.035153303,"threshold_uncertainty_score":0.5292006},"labels":[],"label_agreement":null},{"id":"W4413304054","doi":"10.1080/0969594x.2025.2549158","title":"Diversity: a necessary imperative for assessment research","year":2025,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Evaluation and Performance Assessment","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Diversity (politics); Data science; Engineering ethics; Management science; Computer science; Sociology; Engineering; Anthropology","score_opus":0.4676913733950703,"score_gpt":0.6664621529759441,"score_spread":0.19877077958087386,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413304054","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015077692,0.026355432,0.24268483,0.58491904,0.005387601,0.000952452,0.00036090234,0.0003147819,0.12394723],"genre_scores_gemma":[0.6986411,0.012804275,0.19350605,0.07468904,0.009057436,0.0034905726,0.0003067446,0.000285267,0.0072194203],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.7626415,0.16876954,0.010332136,0.014546841,0.03964183,0.0040681614],"domain_scores_gemma":[0.5476184,0.32694602,0.0122786565,0.061168335,0.039629165,0.012359358],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.21554662,0.00087435934,0.0035201353,0.004174851,0.010639528,0.019614048,0.0039335852,0.008013582,0.0044089537],"category_scores_gemma":[0.331499,0.0014291424,0.00092994963,0.0032466918,0.057139,0.031411104,0.023931507,0.01633723,0.0020139876],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000716005,0.00008324424,0.001847639,0.0007265294,0.00007725163,0.000093961324,0.00928052,0.00053363596,0.0003368888,0.9326436,0.010052593,0.044252455],"study_design_scores_gemma":[0.000032316242,0.000051927986,0.0005294493,0.0008889849,0.000014863423,0.00018146042,0.0020912776,0.0003688725,0.00016799849,0.95496637,0.040671702,0.000034823956],"about_ca_topic_score_codex":0.0023649312,"about_ca_topic_score_gemma":0.002009641,"teacher_disagreement_score":0.7844534,"about_ca_system_score_codex":0.005877886,"about_ca_system_score_gemma":0.023093862,"threshold_uncertainty_score":0.9673707},"labels":[],"label_agreement":null},{"id":"W4415029048","doi":"10.1080/0969594x.2025.2573160","title":"The importance of community in assessment","year":2025,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Education Systems and Policy","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Set (abstract data type); Work (physics)","score_opus":0.08192554165819252,"score_gpt":0.5204617613953847,"score_spread":0.43853621973719215,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415029048","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.00029810038,0.010559529,0.00076974236,0.8943789,0.09078372,0.000011634294,0.00001108876,0.000017380073,0.0031698018],"genre_scores_gemma":[0.07949885,0.030119339,0.0037209874,0.575237,0.2962472,0.00015760881,0.000040622657,0.0002320713,0.014746423],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.90488684,0.054482467,0.006006255,0.005858261,0.024552595,0.004213592],"domain_scores_gemma":[0.63689375,0.2623979,0.0072404384,0.007103111,0.064413145,0.021951564],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.074372515,0.00074694556,0.0019757824,0.0024351012,0.0117925815,0.031833407,0.0046392027,0.024165228,0.0049247066],"category_scores_gemma":[0.25794592,0.00065459026,0.0011869548,0.002028221,0.024977863,0.030000282,0.010781821,0.040325698,0.0010339115],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000030964693,0.000020908366,0.00041195468,0.00057915744,0.000036268157,0.00014255855,0.0032405125,0.00013068714,0.00003643331,0.10529142,0.8451005,0.04497858],"study_design_scores_gemma":[0.000024008415,0.000034618606,0.001099556,0.0027400411,0.000035454636,0.00020457583,0.0037977304,0.0004679606,0.00015163948,0.10143034,0.8899299,0.0000841206],"about_ca_topic_score_codex":0.012386071,"about_ca_topic_score_gemma":0.019184228,"teacher_disagreement_score":0.074372515,"about_ca_system_score_codex":0.011630242,"about_ca_system_score_gemma":0.024175951,"threshold_uncertainty_score":0.39332414},"labels":[],"label_agreement":null},{"id":"W4416894890","doi":"10.1080/0969594x.2025.2591284","title":"An implementation of argument-based validation for assessing college major preferences with a hybrid of Likert-rating and forced-choice formats","year":2025,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Data collection; Measure (data warehouse); Key (lock); Quality (philosophy)","score_opus":0.04844669177838798,"score_gpt":0.4835110550125784,"score_spread":0.4350643632341904,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416894890","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21599603,0.00012250993,0.7313499,0.0013783247,0.0007145423,0.026451118,0.0012544052,0.0015255238,0.021207625],"genre_scores_gemma":[0.23601893,0.00007108289,0.71895576,0.00060981396,0.00007989855,0.041035693,0.0007042863,0.0002880181,0.0022365637],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.72703624,0.21878016,0.017902903,0.007827743,0.02615341,0.0022996147],"domain_scores_gemma":[0.40942302,0.4311151,0.018199928,0.060468394,0.079038076,0.0017555339],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.22691339,0.0016696452,0.0014944646,0.0046746642,0.0027724642,0.0036013266,0.0023836894,0.0021085467,0.004780988],"category_scores_gemma":[0.39466777,0.0013499625,0.0023270715,0.003256688,0.0031252836,0.004003254,0.0049858997,0.0039498163,0.0019102168],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004052783,0.007814126,0.139231,0.001901658,0.00061423896,0.00040253816,0.04054982,0.005708607,0.030399706,0.04781372,0.016594965,0.70491695],"study_design_scores_gemma":[0.0043577827,0.017031813,0.25684407,0.004184096,0.00074815156,0.0023050483,0.029330783,0.24406256,0.16880952,0.11953221,0.15073615,0.0020577596],"about_ca_topic_score_codex":0.0008998437,"about_ca_topic_score_gemma":0.0022168874,"teacher_disagreement_score":0.22691339,"about_ca_system_score_codex":0.0028008313,"about_ca_system_score_gemma":0.00545352,"threshold_uncertainty_score":0.9533534},"labels":[],"label_agreement":null},{"id":"W4417447051","doi":"10.1080/0969594x.2025.2602452","title":"The validity of the assessment for learning measurement instrument for Ethiopian middle school mathematics teachers","year":2025,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Psychometric Methodologies and Testing","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Measure (data warehouse); Reliability (semiconductor); Test validity; Test (biology)","score_opus":0.596064082416965,"score_gpt":0.5664535415641934,"score_spread":0.029610540852771594,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417447051","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98359144,0.0005334151,0.007943384,0.00041254447,0.0000697621,0.00030156705,0.00032956596,0.000035385856,0.006782939],"genre_scores_gemma":[0.9832169,0.00026972312,0.015496327,0.000083182,0.000016351914,0.00034556555,0.00019465388,0.000008955735,0.00036821413],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98673546,0.007481258,0.0013467516,0.0007289325,0.0031503658,0.00055731833],"domain_scores_gemma":[0.94508433,0.035021864,0.0069348225,0.0028796871,0.009147325,0.00093200384],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.025091186,0.00047538828,0.0004735617,0.0027132484,0.0012063328,0.0018636918,0.00080742355,0.00044046834,0.00087939005],"category_scores_gemma":[0.06866992,0.00026786205,0.00042151604,0.0025948011,0.0012362807,0.0013885907,0.001523689,0.00078114384,0.00027314606],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002035547,0.00034412494,0.77201736,0.00037139258,0.000116542324,0.0001514893,0.009709666,0.0010805236,0.0016594401,0.0038474235,0.00094973465,0.20954867],"study_design_scores_gemma":[0.00008673143,0.0011992697,0.94086677,0.0011724074,0.00011704361,0.00058121234,0.02479991,0.00885389,0.0037716462,0.004806842,0.013603433,0.00014079458],"about_ca_topic_score_codex":0.003937587,"about_ca_topic_score_gemma":0.004312937,"teacher_disagreement_score":0.025091186,"about_ca_system_score_codex":0.0019884687,"about_ca_system_score_gemma":0.0040065753,"threshold_uncertainty_score":0.13269645},"labels":[],"label_agreement":null},{"id":"W7118245466","doi":"10.1080/0969594x.2025.2607856","title":"Assessment quality","year":2025,"lang":"en","type":"article","venue":"Assessment in Education Principles Policy and Practice","topic":"Clinical Laboratory Practices and Quality Control","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Quality (philosophy); Quality assessment; Educational assessment; Standards-based assessment; Risk assessment","score_opus":0.09899369132097026,"score_gpt":0.5616235891664735,"score_spread":0.46262989784550324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7118245466","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04233078,0.054186966,0.39004442,0.08648247,0.012557011,0.03622507,0.060342595,0.0048963083,0.31293446],"genre_scores_gemma":[0.5798776,0.022022488,0.24740383,0.024279917,0.0048646913,0.035805784,0.038792267,0.00305727,0.043896172],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.49256623,0.20229204,0.10400222,0.028417872,0.16615167,0.0065699196],"domain_scores_gemma":[0.2101776,0.25141686,0.05892863,0.10508287,0.36365306,0.010740909],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.3105044,0.000925481,0.003634484,0.01309789,0.0036128706,0.013259313,0.00449139,0.0019397994,0.038874157],"category_scores_gemma":[0.7017449,0.0010511371,0.0044008526,0.015090023,0.004153308,0.007958222,0.009859078,0.004695031,0.006956594],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00094833423,0.00019180945,0.04966017,0.0121533135,0.0011883392,0.00008937832,0.0055737603,0.0014423494,0.00049337867,0.04421252,0.119512394,0.7645343],"study_design_scores_gemma":[0.0005024382,0.00053202873,0.07816793,0.02354003,0.0011825031,0.0006052424,0.002414229,0.0033339467,0.002630342,0.04081749,0.84595215,0.0003216537],"about_ca_topic_score_codex":0.01794302,"about_ca_topic_score_gemma":0.012646365,"teacher_disagreement_score":0.3105044,"about_ca_system_score_codex":0.016673435,"about_ca_system_score_gemma":0.051722102,"threshold_uncertainty_score":0.8502708},"labels":[],"label_agreement":null}]}